diff --git a/main/manager-api/src/main/resources/db/changelog/202508200911.sql b/main/manager-api/src/main/resources/db/changelog/202508200911.sql new file mode 100644 index 00000000..9ecc64df --- /dev/null +++ b/main/manager-api/src/main/resources/db/changelog/202508200911.sql @@ -0,0 +1,33 @@ +-- 添加 MinimaxHTTPStream 流式 TTS 供应器 +delete from `ai_model_provider` where id = 'SYSTEM_TTS_MinimaxStreamTTS'; +INSERT INTO `ai_model_provider` (`id`, `model_type`, `provider_code`, `name`, `fields`, `sort`, `creator`, `create_date`, `updater`, `update_date`) VALUES +('SYSTEM_TTS_MinimaxStreamTTS', 'TTS', 'minimax_stream', 'Minimax流式语音合成', '[{"key":"group_id","label":"组ID","type":"string"},{"key":"api_key","label":"API密钥","type":"string"},{"key":"model","label":"模型","type":"string"},{"key":"voice_id","label":"音色ID","type":"string"},{"key":"output_dir","label":"输出目录","type":"string"},{"key":"voice_setting","label":"音色设置","type":"dict","dict_name":"voice_setting"},{"key":"pronunciation_dict","label":"发音字典","type":"dict","dict_name":"pronunciation_dict"},{"key":"audio_setting","label":"音频设置","type":"dict","dict_name":"audio_setting"},{"key":"timber_weights","label":"音色权重","type":"string"}]', 18, 1, NOW(), 1, NOW()); + +-- 添加Minimax流式TTS模型配置 +delete from `ai_model_config` where id = 'TTS_MinimaxStreamTTS'; +INSERT INTO `ai_model_config` VALUES ('TTS_MinimaxStreamTTS', 'TTS', 'MinimaxStreamTTS', 'Minimax流式语音合成', 0, 1, '{"type": "minimax_stream", "group_id": "", "api_key": "", "model": "speech-01", "voice_id": "female-shaonv", "output_dir": "tmp/", "voice_setting": {"speed": 1, "vol": 1, "pitch": 0, "emotion": "happy"}, "pronunciation_dict": {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]}, "audio_setting": {"sample_rate": 24000, "bitrate": 128000, "format": "pcm", "channel": 1}}', NULL, NULL, 21, NULL, NULL, NULL, NULL); + +-- 更新Minimax流式TTS配置说明 +UPDATE `ai_model_config` SET +`doc_link` = 'https://platform.minimaxi.com/', +`remark` = 'Minimax流式TTS配置说明: +1. 需要先申请Minimax API Key +2. 需要填写Group ID +3. 支持多种音色设置和音频参数调整 +4. 支持实时流式合成,具有较低的延迟 +5. 支持自定义发音字典和音色权重 +6. 隐藏参数配置:声音设定(voice_setting)、发音字典(pronunciation_dict)、音色权重(timber_weights) + - 语速(speed): 范围[0.5,2],默认1.0,取值越大语速越快 + - 音量(vol): 范围(0,10],默认1.0,取值越大音量越高 + - 音调(pitch): 范围[-12,12],默认0,取值需为整数 + - timbre_weights与voice_id二选一必填 + - voice_id(请求的音色id,须和weight参数同步填写) + - weight(权重,最多支持4种音色混合。范围[1,100]) +' WHERE `id` = 'TTS_MinimaxStreamTTS'; + +-- 添加Minimax流式TTS音色 +delete from `ai_tts_voice` where tts_model_id = 'TTS_MinimaxStreamTTS'; +-- 默认音色 +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0001', 'TTS_MinimaxStreamTTS', '少女音', 'female-shaonv', '中文', NULL, NULL, NULL, NULL, 1, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0002', 'TTS_MinimaxStreamTTS', '成熟女声', 'female-chengshu', '中文', NULL, NULL, NULL, NULL, 2, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0003', 'TTS_MinimaxStreamTTS', '少年音', 'male-shaonian', '中文', NULL, NULL, NULL, NULL, 3, NULL, NULL, NULL, NULL); \ No newline at end of file diff --git a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml index 0e8d3dc3..86102570 100755 --- a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml +++ b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml @@ -303,3 +303,10 @@ databaseChangeLog: - sqlFile: encoding: utf8 path: classpath:db/changelog/202508131557.sql + - changeSet: + id: 202508200911 + author: RanChen + changes: + - sqlFile: + encoding: utf8 + path: classpath:db/changelog/202508200911.sql diff --git a/main/xiaozhi-server/config.yaml b/main/xiaozhi-server/config.yaml index b3c1b1a8..f6275306 100644 --- a/main/xiaozhi-server/config.yaml +++ b/main/xiaozhi-server/config.yaml @@ -719,26 +719,38 @@ TTS: # voice_id: female-shaonv # weight: 1 # language_boost: auto - -# MinimaxTTSHTTPStream和MinimaxTTSWebSocketStream还在测试,测试完再开放 -# -# MinimaxTTSHTTPStream: -# # Minimax流式语音合成服务 -# type: minimax_httpstream -# output_dir: tmp/ -# group_id: 你的minimax平台groupID -# api_key: 你的minimax平台接口密钥 -# model: "speech-01-turbo" -# voice_id: "female-shaonv" -# -# MinimaxTTSWebSocketStream: -# type: minimax_webSocket -# output_dir: tmp/ -# group_id: 你的minimax平台groupID -# api_key: 你的minimax平台接口密钥 -# model: "speech-01-turbo" -# voice_id: "female-shaonv" - + MinimaxTTSHTTPStream: + # Minimax流式语音合成服务 + type: minimax_httpstream + output_dir: tmp/ + group_id: 你的minimax平台groupID + api_key: 你的minimax平台接口密钥 + model: "speech-01-turbo" + voice_id: "female-shaonv" + # 以下可不用设置,使用默认设置 + # voice_setting: + # voice_id: "male-qn-qingse" + # speed: 1 + # vol: 1 + # pitch: 0 + # emotion: "happy" + # pronunciation_dict: + # tone: + # - "处理/(chu3)(li3)" + # - "危险/dangerous" + # audio_setting: + # sample_rate: 24000 + # bitrate: 128000 + # format: "mp3" + # channel: 1 + # timber_weights: + # - + # voice_id: male-qn-qingse + # weight: 1 + # - + # voice_id: female-shaonv + # weight: 1 + # language_boost: auto AliyunTTS: # 阿里云智能语音交互服务,需要先在阿里云平台开通服务,然后获取验证信息 # 平台地址:https://nls-portal.console.aliyun.com/ diff --git a/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py b/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py index 605b3efd..be944521 100644 --- a/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py +++ b/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py @@ -1,11 +1,18 @@ import os -import uuid import json -import requests -from datetime import datetime -from typing import Iterator, Optional, Union -from core.providers.tts.base import TTSProviderBase +import queue +import asyncio +import traceback +import aiohttp from core.utils.util import parse_string_to_list +from config.logger import setup_logging +from core.utils.tts import MarkdownCleaner +from core.providers.tts.base import TTSProviderBase +from core.utils import opus_encoder_utils, textUtils +from core.providers.tts.dto.dto import SentenceType, ContentType + +TAG = __name__ +logger = setup_logging() class TTSProvider(TTSProviderBase): @@ -28,9 +35,9 @@ class TTSProvider(TTSProviderBase): } default_pronunciation_dict = {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]} defult_audio_setting = { - "sample_rate": 32000, + "sample_rate": 24000, "bitrate": 128000, - "format": "mp3", + "format": "pcm", "channel": 1, } self.voice_setting = { @@ -47,66 +54,108 @@ class TTSProvider(TTSProviderBase): if self.voice: self.voice_setting["voice_id"] = self.voice - self.host = "api.minimax.chat" + self.host = "api.minimaxi.com" # 备用地址:api-bj.minimaxi.com self.api_url = f"https://{self.host}/v1/t2a_v2?GroupId={self.group_id}" self.header = { "Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}", } - self.audio_file_type = defult_audio_setting.get("format", "mp3") + self.audio_file_type = defult_audio_setting.get("format", "pcm") - def generate_filename(self, extension=".mp3"): - return os.path.join( - self.output_file, - f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}", + self.opus_encoder = opus_encoder_utils.OpusEncoderUtils( + sample_rate=24000, channels=1, frame_size_ms=60 ) - async def text_to_speak(self, text, output_file): - """非流式语音合成(保留原有实现)""" - request_json = { - "model": self.model, - "text": text, - "stream": False, - "voice_setting": self.voice_setting, - "pronunciation_dict": self.pronunciation_dict, - "audio_setting": self.audio_setting, - } + # PCM缓冲区 + self.pcm_buffer = bytearray() - if type(self.timber_weights) is list and len(self.timber_weights) > 0: - request_json["timber_weights"] = self.timber_weights - request_json["voice_setting"]["voice_id"] = "" + def tts_text_priority_thread(self): + """流式文本处理线程""" + while not self.conn.stop_event.is_set(): + try: + message = self.tts_text_queue.get(timeout=1) + if message.sentence_type == SentenceType.FIRST: + # 初始化参数 + self.tts_stop_request = False + self.processed_chars = 0 + self.tts_text_buff = [] + self.before_stop_play_files.clear() + self.reset_flow_controller() + elif ContentType.TEXT == message.content_type: + self.tts_text_buff.append(message.content_detail) + segment_text = self._get_segment_text() + if segment_text: + self.to_tts_single_stream(segment_text) - try: - resp = requests.post( - self.api_url, json.dumps(request_json), headers=self.header - ) - if resp.json()["base_resp"]["status_code"] == 0: - data = resp.json()["data"]["audio"] - audio_bytes = bytes.fromhex(data) - if output_file: - with open(output_file, "wb") as file_to_save: - file_to_save.write(audio_bytes) - else: - return audio_bytes + elif ContentType.FILE == message.content_type: + logger.bind(tag=TAG).info( + f"添加音频文件到待播放列表: {message.content_file}" + ) + if message.content_file and os.path.exists(message.content_file): + # 先处理文件音频数据 + self._process_audio_file_stream( + message.content_file, + callback=lambda audio_data: self.handle_audio_file( + audio_data, message.content_detail + ), + ) + + if message.sentence_type == SentenceType.LAST: + # 处理剩余的文本 + self._process_remaining_text_stream(True) + + except queue.Empty: + continue + except Exception as e: + logger.bind(tag=TAG).error( + f"处理TTS文本失败: {str(e)}, 类型: {type(e).__name__}, 堆栈: {traceback.format_exc()}" + ) + + def _process_remaining_text_stream(self, is_last=False): + """处理剩余的文本并生成语音 + Returns: + bool: 是否成功处理了文本 + """ + full_text = "".join(self.tts_text_buff) + remaining_text = full_text[self.processed_chars :] + if remaining_text: + segment_text = textUtils.get_string_no_punctuation_or_emoji(remaining_text) + if segment_text: + self.to_tts_single_stream(segment_text, is_last) + self.processed_chars += len(full_text) else: - raise Exception( - f"{__name__} status_code: {resp.status_code} response: {resp.content}" + self._process_before_stop_play_files() + else: + self._process_before_stop_play_files() + + def to_tts_single_stream(self, text, is_last=False): + try: + max_repeat_time = 5 + text = MarkdownCleaner.clean_markdown(text) + try: + asyncio.run(self.text_to_speak(text, is_last)) + except Exception as e: + logger.bind(tag=TAG).warning( + f"语音生成失败{5 - max_repeat_time + 1}次: {text},错误: {e}" + ) + max_repeat_time -= 1 + + if max_repeat_time > 0: + logger.bind(tag=TAG).info( + f"语音生成成功: {text},重试{5 - max_repeat_time}次" + ) + else: + logger.bind(tag=TAG).error( + f"语音生成失败: {text},请检查网络或服务是否正常" ) except Exception as e: - raise Exception(f"{__name__} error: {e}") + logger.bind(tag=TAG).error(f"Failed to generate TTS file: {e}") + finally: + return None - def text_to_speak_stream( - self, - text: str, - chunk_callback: Optional[callable] = None - ) -> Iterator[bytes]: - """ - 流式语音合成方法 - :param text: 要合成的文本 - :param chunk_callback: 可选的回调函数,用于处理每个音频块 - :return: 生成器,每次产生一个音频数据块(bytes) - """ - request_json = { + async def text_to_speak(self, text, is_last): + """流式处理TTS音频,每句只推送一次音频列表""" + payload = { "model": self.model, "text": text, "stream": True, @@ -115,116 +164,98 @@ class TTSProvider(TTSProviderBase): "audio_setting": self.audio_setting, } - if isinstance(self.timber_weights, list) and len(self.timber_weights) > 0: - request_json["timber_weights"] = self.timber_weights - request_json["voice_setting"]["voice_id"] = "" + if type(self.timber_weights) is list and len(self.timber_weights) > 0: + payload["timber_weights"] = self.timber_weights + payload["voice_setting"]["voice_id"] = "" + frame_bytes = int( + self.opus_encoder.sample_rate + * self.opus_encoder.channels # 1 + * self.opus_encoder.frame_size_ms + / 1000 + * 2 + ) # 16-bit = 2 bytes try: - with requests.post( - self.api_url, - data=json.dumps(request_json), - headers=self.header, - stream=True - ) as response: - - # 检查HTTP状态码 - if response.status_code != 200: - raise Exception( - f"HTTP error: {response.status_code}, response: {response.text}" - ) - - # 处理流式响应 - for line in response.iter_lines(): - if line: # 过滤空行 - # 检查是否为数据行 (SSE格式) - if line.startswith(b'data:'): + async with aiohttp.ClientSession() as session: + async with session.post( + self.api_url, + headers=self.header, + data=json.dumps(payload), + timeout=10, + ) as resp: + + if resp.status != 200: + logger.bind(tag=TAG).error( + f"TTS请求失败: {resp.status}, {await resp.text()}" + ) + self.tts_audio_queue.put((SentenceType.LAST, [], None)) + return + + self.pcm_buffer.clear() + self.tts_audio_queue.put((SentenceType.FIRST, [], text)) + + # 处理音频流数据 + buffer = b"" + async for chunk in resp.content.iter_any(): + if not chunk: + continue + + buffer += chunk + while True: + # 查找数据块分隔符 + header_pos = buffer.find(b"data: ") + if header_pos == -1: + break + + end_pos = buffer.find(b"\n\n", header_pos) + if end_pos == -1: + break + + # 提取单个完整JSON块 + json_str = buffer[header_pos + 6 : end_pos].decode("utf-8") + buffer = buffer[end_pos + 2 :] + try: - data = json.loads(line[5:].strip()) # 去掉"data:"前缀 - - # 检查API状态码 - if data.get("base_resp", {}).get("status_code", -1) != 0: - raise Exception( - f"API error: {data.get('base_resp', {}).get('status_msg')}" - ) - - # 跳过非音频数据块 - if "extra_info" in data: - continue - - # 提取音频数据 + data = json.loads(json_str) + status = data.get("data", {}).get("status", 1) audio_hex = data.get("data", {}).get("audio") - if audio_hex: - audio_chunk = bytes.fromhex(audio_hex) - if chunk_callback: - chunk_callback(audio_chunk) - yield audio_chunk - - except json.JSONDecodeError: - # 忽略JSON解析错误(可能是心跳包等) + + # 仅处理status=1的有效音频块 忽略status=2的结束汇总块 + if status == 1 and audio_hex: + pcm_data = bytes.fromhex(audio_hex) + self.pcm_buffer.extend(pcm_data) + + except json.JSONDecodeError as e: + logger.bind(tag=TAG).error(f"JSON解析失败: {e}") continue - except Exception as e: - raise e - - except Exception as e: - raise Exception(f"{__name__} stream error: {e}") - def save_stream_to_file( - self, - text: str, - output_file: Optional[str] = None, - progress_callback: Optional[callable] = None - ) -> str: - """ - 流式合成并保存到文件 - :param text: 要合成的文本 - :param output_file: 输出文件路径,如果为None则自动生成 - :param progress_callback: 可选的回调函数,接收已写入的字节数 - :return: 保存的文件路径 - """ - if not output_file: - output_file = self.generate_filename(extension=f".{self.audio_file_type}") - - os.makedirs(os.path.dirname(output_file), exist_ok=True) - - total_bytes = 0 - try: - with open(output_file, "wb") as audio_file: - for audio_chunk in self.text_to_speak_stream(text): - audio_file.write(audio_chunk) - audio_file.flush() - total_bytes += len(audio_chunk) - if progress_callback: - progress_callback(total_bytes) - return output_file - except Exception as e: - # 清理可能创建的不完整文件 - if os.path.exists(output_file): - os.remove(output_file) - raise e + while len(self.pcm_buffer) >= frame_bytes: + frame = bytes(self.pcm_buffer[:frame_bytes]) + del self.pcm_buffer[:frame_bytes] + + self.opus_encoder.encode_pcm_to_opus_stream( + frame, end_of_stream=False, callback=self.handle_opus + ) + + # flush 剩余不足一帧的数据 + if self.pcm_buffer: + self.opus_encoder.encode_pcm_to_opus_stream( + bytes(self.pcm_buffer), + end_of_stream=True, + callback=self.handle_opus, + ) + self.pcm_buffer.clear() + + # 如果是最后一段,输出音频获取完毕 + if is_last: + self._process_before_stop_play_files() - def stream_to_audio_player(self, text: str, player_command: list = None): - """ - 流式合成并直接播放音频 - :param text: 要合成的文本 - :param player_command: 音频播放器命令,默认使用mpv - """ - if player_command is None: - player_command = ["mpv", "--no-cache", "--no-terminal", "--", "fd://0"] - - try: - import subprocess - player_process = subprocess.Popen( - player_command, - stdin=subprocess.PIPE, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - ) - - for audio_chunk in self.text_to_speak_stream(text): - player_process.stdin.write(audio_chunk) - player_process.stdin.flush() - - player_process.stdin.close() - player_process.wait() except Exception as e: - raise Exception(f"Audio player error: {e}") \ No newline at end of file + logger.bind(tag=TAG).error(f"TTS请求异常: {e}") + self.tts_audio_queue.put((SentenceType.LAST, [], None)) + + async def close(self): + """资源清理""" + await super().close() + if hasattr(self, "opus_encoder"): + self.opus_encoder.close() diff --git a/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py b/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py deleted file mode 100644 index bda6b47f..00000000 --- a/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py +++ /dev/null @@ -1,180 +0,0 @@ -import os -import uuid -import json -import asyncio -import websockets -import ssl -from datetime import datetime -from core.providers.tts.base import TTSProviderBase -from core.utils.util import parse_string_to_list - - -class TTSProvider(TTSProviderBase): - def __init__(self, config, delete_audio_file): - super().__init__(config, delete_audio_file) - self.group_id = config.get("group_id") - self.api_key = config.get("api_key") - self.model = config.get("model") - - # 初始化语音设置 - default_voice_setting = { - "voice_id": "female-shaonv", - "speed": 1, - "vol": 1, - "pitch": 0, - "emotion": "happy", - } - default_pronunciation_dict = {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]} - default_audio_setting = { - "sample_rate": 32000, - "bitrate": 128000, - "format": "mp3", - "channel": 1, - } - - # 合并配置 - self.voice_setting = { - **default_voice_setting, - **config.get("voice_setting", {}), - } - self.pronunciation_dict = { - **default_pronunciation_dict, - **config.get("pronunciation_dict", {}), - } - self.audio_setting = { - **default_audio_setting, - **config.get("audio_setting", {}) - } - self.timber_weights = parse_string_to_list(config.get("timber_weights")) - - # 设置语音ID - if config.get("private_voice"): - self.voice_setting["voice_id"] = config.get("private_voice") - elif config.get("voice_id"): - self.voice_setting["voice_id"] = config.get("voice_id") - - # WebSocket配置 - self.ws_url = "wss://api.minimaxi.com/ws/v1/t2a_v2" - self.headers = { - "Authorization": f"Bearer {self.api_key}", - "GroupId": self.group_id - } - self.audio_file_type = self.audio_setting.get("format", "mp3") - - def generate_filename(self, extension=".mp3"): - """生成唯一的音频文件名""" - return os.path.join( - self.output_file, - f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}", - ) - - async def _establish_connection(self): - """建立WebSocket连接""" - ssl_context = ssl.create_default_context() - ssl_context.check_hostname = False - ssl_context.verify_mode = ssl.CERT_NONE - - try: - ws = await websockets.connect( - self.ws_url, - additional_headers=self.headers, - ssl=ssl_context - ) - connected = json.loads(await ws.recv()) - if connected.get("event") == "connected_success": - print("连接成功") - return ws - return None - except Exception as e: - print(f"连接失败: {e}") - return None - - async def _start_task(self, websocket): - """发送任务开始请求""" - start_msg = { - "event": "task_start", - "model": self.model, - "voice_setting": self.voice_setting, - "pronunciation_dict": self.pronunciation_dict, - "audio_setting": self.audio_setting - } - - if self.timber_weights and len(self.timber_weights) > 0: - start_msg["timber_weights"] = self.timber_weights - start_msg["voice_setting"]["voice_id"] = "" - - await websocket.send(json.dumps(start_msg)) - response = json.loads(await websocket.recv()) - return response.get("event") == "task_started" - - async def _continue_task(self, websocket, text): - """发送继续请求并收集音频数据""" - await websocket.send(json.dumps({ - "event": "task_continue", - "text": text - })) - - audio_chunks = [] - while True: - response = json.loads(await websocket.recv()) - if "data" in response and "audio" in response["data"]: - audio_chunks.append(response["data"]["audio"]) - if response.get("is_final"): - break - return "".join(audio_chunks) - - async def _close_connection(self, websocket): - """关闭连接""" - if websocket: - await websocket.send(json.dumps({"event": "task_finish"})) - await websocket.close() - print("连接已关闭") - - async def text_to_speak(self, text, output_file=None): - """主方法:文本转语音""" - ws = await self._establish_connection() - if not ws: - raise Exception("无法建立WebSocket连接") - - try: - if not await self._start_task(ws): - raise Exception("任务启动失败") - - hex_audio = await self._continue_task(ws, text) - audio_bytes = bytes.fromhex(hex_audio) - - # 保存到文件或返回二进制数据 - if output_file: - with open(output_file, "wb") as f: - f.write(audio_bytes) - print(f"音频已保存为{output_file}") - return output_file - else: - # 返回音频二进制数据(不播放) - return audio_bytes - - finally: - await self._close_connection(ws) - - -async def main(): - """测试用主函数""" - # 示例配置 - config = { - "group_id": "YOUR_GROUP_ID", # 替换为实际的group_id - "api_key": "YOUR_API_KEY", # 替换为实际的api_key - "model": "your-model", # 替换为实际的模型名称 - "voice_id": "male-qn-qingse", - "voice_setting": { - "speed": 1.2, - "emotion": "happy" - } - } - - tts = TTSProvider(config, delete_audio_file=True) - output_file = tts.generate_filename() - await tts.text_to_speak("这是一个测试文本,用于验证流式语音合成功能", output_file) - - -if __name__ == "__main__": - asyncio.run(main()) \ No newline at end of file