diff --git a/main/manager-api/src/main/resources/db/changelog/202509051745.sql b/main/manager-api/src/main/resources/db/changelog/202509051745.sql new file mode 100644 index 00000000..6dc92222 --- /dev/null +++ b/main/manager-api/src/main/resources/db/changelog/202509051745.sql @@ -0,0 +1,45 @@ +-- 添加 MinimaxHTTPStream 流式 TTS 供应器 +delete from `ai_model_provider` where id = 'SYSTEM_TTS_MinimaxStreamTTS'; +INSERT INTO `ai_model_provider` (`id`, `model_type`, `provider_code`, `name`, `fields`, `sort`, `creator`, `create_date`, `updater`, `update_date`) VALUES +('SYSTEM_TTS_MinimaxStreamTTS', 'TTS', 'minimax_httpstream', 'Minimax流式语音合成', '[{"key":"group_id","label":"组ID","type":"string"},{"key":"api_key","label":"API密钥","type":"string"},{"key":"model","label":"模型","type":"string"},{"key":"voice_id","label":"音色ID","type":"string"},{"key":"output_dir","label":"输出目录","type":"string"},{"key":"voice_setting","label":"音色设置","type":"dict","dict_name":"voice_setting"},{"key":"pronunciation_dict","label":"发音字典","type":"dict","dict_name":"pronunciation_dict"},{"key":"audio_setting","label":"音频设置","type":"dict","dict_name":"audio_setting"},{"key":"timber_weights","label":"音色权重","type":"string"}]', 18, 1, NOW(), 1, NOW()); + +-- 添加Minimax流式TTS模型配置 +delete from `ai_model_config` where id = 'TTS_MinimaxStreamTTS'; +INSERT INTO `ai_model_config` VALUES ('TTS_MinimaxStreamTTS', 'TTS', 'MinimaxStreamTTS', 'Minimax流式语音合成', 0, 1, '{"type": "minimax_httpstream", "group_id": "", "api_key": "", "model": "speech-01-turbo", "voice_id": "female-shaonv", "output_dir": "tmp/", "voice_setting": {"speed": 1, "vol": 1, "pitch": 0, "emotion": "happy"}, "pronunciation_dict": {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]}, "audio_setting": {"sample_rate": 24000, "bitrate": 128000, "format": "pcm", "channel": 1}}', NULL, NULL, 21, NULL, NULL, NULL, NULL); + +-- 更新Minimax流式TTS配置说明 +UPDATE `ai_model_config` SET +`doc_link` = 'https://platform.minimaxi.com/', +`remark` = 'Minimax流式TTS配置说明: +1. 需要先申请Minimax API Key +2. 需要填写Group ID +3. 支持多种音色设置和音频参数调整 +4. 支持实时流式合成,具有较低的延迟 +5. 支持自定义发音字典和音色权重 +6. 隐藏参数配置:声音设定(voice_setting)、发音字典(pronunciation_dict)、音色权重(timber_weights) + - 语速(speed): 范围[0.5,2],默认1.0,取值越大语速越快 + - 音量(vol): 范围(0,10],默认1.0,取值越大音量越高 + - 音调(pitch): 范围[-12,12],默认0,取值需为整数 + - 情绪(emotion): 控制合成语音的情绪,支持7种值:["happy", "sad", "angry", "fearful", "disgusted", "surprised", "calm"],该参数仅对 speech-2.5-hd-preview、speech-2.5-turbo-preview、speech-02-hd、speech-02-turbo、speech-01-turbo、speech-01-hd 生效 + - timbre_weights与voice_id二选一必填 + - voice_id(请求的音色id,须和weight参数同步填写) + - weight(权重,最多支持4种音色混合。范围[1,100]) +' WHERE `id` = 'TTS_MinimaxStreamTTS'; + +-- 添加Minimax流式TTS音色 +delete from `ai_tts_voice` where tts_model_id = 'TTS_MinimaxStreamTTS'; + +-- 默认音色 +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0001', 'TTS_MinimaxStreamTTS', '少女音', 'female-shaonv', '中文', NULL, NULL, NULL, NULL, 1, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0002', 'TTS_MinimaxStreamTTS', '成熟女声', 'female-chengshu', '中文', NULL, NULL, NULL, NULL, 2, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0003', 'TTS_MinimaxStreamTTS', '霸道少爷', 'badao_shaoye', '中文', NULL, NULL, NULL, NULL, 3, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0004', 'TTS_MinimaxStreamTTS', '病娇弟弟', 'bingjiao_didi', '中文', NULL, NULL, NULL, NULL, 4, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0005', 'TTS_MinimaxStreamTTS', '纯真学弟', 'chunzhen_xuedi', '中文', NULL, NULL, NULL, NULL, 5, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0006', 'TTS_MinimaxStreamTTS', '冷淡学长', 'lengdan_xiongzhang', '中文', NULL, NULL, NULL, NULL, 6, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0007', 'TTS_MinimaxStreamTTS', '甜美小玲', 'tianxin_xiaoling', '中文', NULL, NULL, NULL, NULL, 7, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0008', 'TTS_MinimaxStreamTTS', '俏皮萌妹', 'qiaopi_mengmei', '中文', NULL, NULL, NULL, NULL, 8, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0009', 'TTS_MinimaxStreamTTS', '妩媚御姐', 'wumei_yujie', '中文', NULL, NULL, NULL, NULL, 9, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0010', 'TTS_MinimaxStreamTTS', '嗲嗲学妹', 'diadia_xuemei', '中文', NULL, NULL, NULL, NULL, 7, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0011', 'TTS_MinimaxStreamTTS', '淡雅学姐', 'danya_xuejie', '中文', NULL, NULL, NULL, NULL, 8, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0012', 'TTS_MinimaxStreamTTS', 'Santa Claus', 'Santa_Claus', '中文', NULL, NULL, NULL, NULL, 9, NULL, NULL, NULL, NULL); +INSERT INTO `ai_tts_voice` VALUES ('TTS_MinimaxStreamTTS_0013', 'TTS_MinimaxStreamTTS', 'Grinch', 'Grinch', '中文', NULL, NULL, NULL, NULL, 10, NULL, NULL, NULL, NULL); diff --git a/main/manager-api/src/main/resources/db/changelog/202509091042.sql b/main/manager-api/src/main/resources/db/changelog/202509091042.sql new file mode 100644 index 00000000..5392738a --- /dev/null +++ b/main/manager-api/src/main/resources/db/changelog/202509091042.sql @@ -0,0 +1,10 @@ +-- 删除非流式MiniMax TTS配置,保留流式版本 + +-- 删除旧的非流式MiniMax TTS模型配置 +DELETE FROM `ai_model_config` WHERE `id` = 'TTS_MinimaxTTS'; + +-- 删除旧的非流式MiniMax TTS供应器配置 +DELETE FROM `ai_model_provider` WHERE `id` = 'SYSTEM_TTS_minimax'; + +-- 删除旧的非流式MiniMax TTS音色配置 +DELETE FROM `ai_tts_voice` WHERE `tts_model_id` = 'TTS_MinimaxTTS'; diff --git a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml index 817281be..e7b19f0d 100755 --- a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml +++ b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml @@ -303,6 +303,20 @@ databaseChangeLog: - sqlFile: encoding: utf8 path: classpath:db/changelog/202508131557.sql + - changeSet: + id: 202508271113 + author: cgd + changes: + - sqlFile: + encoding: utf8 + path: classpath:db/changelog/202508271113.sql + - changeSet: + id: 202509051745 + author: RanChen + changes: + - sqlFile: + encoding: utf8 + path: classpath:db/changelog/202509051745.sql - changeSet: id: 202509081140 author: cgd @@ -311,9 +325,9 @@ databaseChangeLog: encoding: utf8 path: classpath:db/changelog/202509081140.sql - changeSet: - id: 202508271113 + id: 202509091042 author: cgd changes: - sqlFile: encoding: utf8 - path: classpath:db/changelog/202508271113.sql \ No newline at end of file + path: classpath:db/changelog/202509091042.sql \ No newline at end of file diff --git a/main/xiaozhi-server/config.yaml b/main/xiaozhi-server/config.yaml index e84cd905..1cfdd964 100644 --- a/main/xiaozhi-server/config.yaml +++ b/main/xiaozhi-server/config.yaml @@ -707,19 +707,13 @@ TTS: inp_refs: [] sample_steps: 32 if_sr: false - MinimaxTTS: - # Minimax语音合成服务,需要先在minimax平台创建账户充值,并获取登录信息 - # 平台地址:https://platform.minimaxi.com/ - # 充值地址:https://platform.minimaxi.com/user-center/payment/balance - # group_id地址:https://platform.minimaxi.com/user-center/basic-information - # api_key地址:https://platform.minimaxi.com/user-center/basic-information/interface-key - # 定义TTS API类型 - type: minimax + MinimaxTTSHTTPStream: + # Minimax流式语音合成服务 + type: minimax_httpstream output_dir: tmp/ group_id: 你的minimax平台groupID api_key: 你的minimax平台接口密钥 model: "speech-01-turbo" - # 此处设置将优先于voice_setting中voice_id的设置;如都不设置,默认为 female-shaonv voice_id: "female-shaonv" # 以下可不用设置,使用默认设置 # voice_setting: @@ -733,7 +727,7 @@ TTS: # - "处理/(chu3)(li3)" # - "危险/dangerous" # audio_setting: - # sample_rate: 32000 + # sample_rate: 24000 # bitrate: 128000 # format: "mp3" # channel: 1 @@ -745,26 +739,6 @@ TTS: # voice_id: female-shaonv # weight: 1 # language_boost: auto - -# MinimaxTTSHTTPStream和MinimaxTTSWebSocketStream还在测试,测试完再开放 -# -# MinimaxTTSHTTPStream: -# # Minimax流式语音合成服务 -# type: minimax_httpstream -# output_dir: tmp/ -# group_id: 你的minimax平台groupID -# api_key: 你的minimax平台接口密钥 -# model: "speech-01-turbo" -# voice_id: "female-shaonv" -# -# MinimaxTTSWebSocketStream: -# type: minimax_webSocket -# output_dir: tmp/ -# group_id: 你的minimax平台groupID -# api_key: 你的minimax平台接口密钥 -# model: "speech-01-turbo" -# voice_id: "female-shaonv" - AliyunTTS: # 阿里云智能语音交互服务,需要先在阿里云平台开通服务,然后获取验证信息 # 平台地址:https://nls-portal.console.aliyun.com/ diff --git a/main/xiaozhi-server/core/providers/tts/minimax.py b/main/xiaozhi-server/core/providers/tts/minimax.py deleted file mode 100644 index 75e25e2e..00000000 --- a/main/xiaozhi-server/core/providers/tts/minimax.py +++ /dev/null @@ -1,95 +0,0 @@ -import os -import uuid -import json -import requests -from datetime import datetime -from core.providers.tts.base import TTSProviderBase -from core.utils.util import parse_string_to_list - - -class TTSProvider(TTSProviderBase): - def __init__(self, config, delete_audio_file): - super().__init__(config, delete_audio_file) - self.group_id = config.get("group_id") - self.api_key = config.get("api_key") - self.model = config.get("model") - if config.get("private_voice"): - self.voice = config.get("private_voice") - else: - self.voice = config.get("voice_id") - - default_voice_setting = { - "voice_id": "female-shaonv", - "speed": 1, - "vol": 1, - "pitch": 0, - "emotion": "happy", - } - default_pronunciation_dict = {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]} - defult_audio_setting = { - "sample_rate": 32000, - "bitrate": 128000, - "format": "mp3", - "channel": 1, - } - self.voice_setting = { - **default_voice_setting, - **config.get("voice_setting", {}), - } - self.pronunciation_dict = { - **default_pronunciation_dict, - **config.get("pronunciation_dict", {}), - } - self.audio_setting = {**defult_audio_setting, **config.get("audio_setting", {})} - self.timber_weights = parse_string_to_list(config.get("timber_weights")) - - if self.voice: - self.voice_setting["voice_id"] = self.voice - - self.host = "api.minimax.chat" - self.api_url = f"https://{self.host}/v1/t2a_v2?GroupId={self.group_id}" - self.header = { - "Content-Type": "application/json", - "Authorization": f"Bearer {self.api_key}", - } - self.audio_file_type = defult_audio_setting.get("format", "mp3") - - def generate_filename(self, extension=".mp3"): - return os.path.join( - self.output_file, - f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}", - ) - - async def text_to_speak(self, text, output_file): - request_json = { - "model": self.model, - "text": text, - "stream": False, - "voice_setting": self.voice_setting, - "pronunciation_dict": self.pronunciation_dict, - "audio_setting": self.audio_setting, - } - - if type(self.timber_weights) is list and len(self.timber_weights) > 0: - request_json["timber_weights"] = self.timber_weights - request_json["voice_setting"]["voice_id"] = "" - - try: - resp = requests.post( - self.api_url, json.dumps(request_json), headers=self.header - ) - # 检查返回请求数据的status_code是否为0 - if resp.json()["base_resp"]["status_code"] == 0: - data = resp.json()["data"]["audio"] - audio_bytes = bytes.fromhex(data) - if output_file: - with open(output_file, "wb") as file_to_save: - file_to_save.write(audio_bytes) - else: - return audio_bytes - else: - raise Exception( - f"{__name__} status_code: {resp.status_code} response: {resp.content}" - ) - except Exception as e: - raise Exception(f"{__name__} error: {e}") diff --git a/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py b/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py index 605b3efd..514d8912 100644 --- a/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py +++ b/main/xiaozhi-server/core/providers/tts/minimax_httpstream.py @@ -1,11 +1,20 @@ import os -import uuid import json +import time +import queue +import asyncio +import aiohttp import requests -from datetime import datetime -from typing import Iterator, Optional, Union -from core.providers.tts.base import TTSProviderBase +import traceback +from config.logger import setup_logging +from core.utils.tts import MarkdownCleaner from core.utils.util import parse_string_to_list +from core.providers.tts.base import TTSProviderBase +from core.utils import opus_encoder_utils, textUtils +from core.providers.tts.dto.dto import SentenceType, ContentType + +TAG = __name__ +logger = setup_logging() class TTSProvider(TTSProviderBase): @@ -28,9 +37,9 @@ class TTSProvider(TTSProviderBase): } default_pronunciation_dict = {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]} defult_audio_setting = { - "sample_rate": 32000, + "sample_rate": 24000, "bitrate": 128000, - "format": "mp3", + "format": "pcm", "channel": 1, } self.voice_setting = { @@ -47,66 +56,101 @@ class TTSProvider(TTSProviderBase): if self.voice: self.voice_setting["voice_id"] = self.voice - self.host = "api.minimax.chat" + self.host = "api.minimaxi.com" # 备用地址:api-bj.minimaxi.com self.api_url = f"https://{self.host}/v1/t2a_v2?GroupId={self.group_id}" self.header = { "Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}", } - self.audio_file_type = defult_audio_setting.get("format", "mp3") + self.audio_file_type = defult_audio_setting.get("format", "pcm") - def generate_filename(self, extension=".mp3"): - return os.path.join( - self.output_file, - f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}", + self.opus_encoder = opus_encoder_utils.OpusEncoderUtils( + sample_rate=24000, channels=1, frame_size_ms=60 ) - async def text_to_speak(self, text, output_file): - """非流式语音合成(保留原有实现)""" - request_json = { - "model": self.model, - "text": text, - "stream": False, - "voice_setting": self.voice_setting, - "pronunciation_dict": self.pronunciation_dict, - "audio_setting": self.audio_setting, - } + # PCM缓冲区 + self.pcm_buffer = bytearray() - if type(self.timber_weights) is list and len(self.timber_weights) > 0: - request_json["timber_weights"] = self.timber_weights - request_json["voice_setting"]["voice_id"] = "" + def tts_text_priority_thread(self): + """流式文本处理线程""" + while not self.conn.stop_event.is_set(): + try: + message = self.tts_text_queue.get(timeout=1) + if message.sentence_type == SentenceType.FIRST: + # 初始化参数 + self.tts_stop_request = False + self.processed_chars = 0 + self.tts_text_buff = [] + self.before_stop_play_files.clear() + elif ContentType.TEXT == message.content_type: + self.tts_text_buff.append(message.content_detail) + segment_text = self._get_segment_text() + if segment_text: + self.to_tts_single_stream(segment_text) - try: - resp = requests.post( - self.api_url, json.dumps(request_json), headers=self.header - ) - if resp.json()["base_resp"]["status_code"] == 0: - data = resp.json()["data"]["audio"] - audio_bytes = bytes.fromhex(data) - if output_file: - with open(output_file, "wb") as file_to_save: - file_to_save.write(audio_bytes) - else: - return audio_bytes + elif ContentType.FILE == message.content_type: + logger.bind(tag=TAG).info( + f"添加音频文件到待播放列表: {message.content_file}" + ) + if message.content_file and os.path.exists(message.content_file): + # 先处理文件音频数据 + self._process_audio_file_stream(message.content_file, callback=lambda audio_data: self.handle_audio_file(audio_data, message.content_detail)) + if message.sentence_type == SentenceType.LAST: + # 处理剩余的文本 + self._process_remaining_text_stream(True) + + except queue.Empty: + continue + except Exception as e: + logger.bind(tag=TAG).error( + f"处理TTS文本失败: {str(e)}, 类型: {type(e).__name__}, 堆栈: {traceback.format_exc()}" + ) + + def _process_remaining_text_stream(self, is_last=False): + """处理剩余的文本并生成语音 + Returns: + bool: 是否成功处理了文本 + """ + full_text = "".join(self.tts_text_buff) + remaining_text = full_text[self.processed_chars :] + if remaining_text: + segment_text = textUtils.get_string_no_punctuation_or_emoji(remaining_text) + if segment_text: + self.to_tts_single_stream(segment_text, is_last) + self.processed_chars += len(full_text) else: - raise Exception( - f"{__name__} status_code: {resp.status_code} response: {resp.content}" + self._process_before_stop_play_files() + else: + self._process_before_stop_play_files() + + def to_tts_single_stream(self, text, is_last=False): + try: + max_repeat_time = 5 + text = MarkdownCleaner.clean_markdown(text) + try: + asyncio.run(self.text_to_speak(text, is_last)) + except Exception as e: + logger.bind(tag=TAG).warning( + f"语音生成失败{5 - max_repeat_time + 1}次: {text},错误: {e}" + ) + max_repeat_time -= 1 + + if max_repeat_time > 0: + logger.bind(tag=TAG).info( + f"语音生成成功: {text},重试{5 - max_repeat_time}次" + ) + else: + logger.bind(tag=TAG).error( + f"语音生成失败: {text},请检查网络或服务是否正常" ) except Exception as e: - raise Exception(f"{__name__} error: {e}") + logger.bind(tag=TAG).error(f"Failed to generate TTS file: {e}") + finally: + return None - def text_to_speak_stream( - self, - text: str, - chunk_callback: Optional[callable] = None - ) -> Iterator[bytes]: - """ - 流式语音合成方法 - :param text: 要合成的文本 - :param chunk_callback: 可选的回调函数,用于处理每个音频块 - :return: 生成器,每次产生一个音频数据块(bytes) - """ - request_json = { + async def text_to_speak(self, text, is_last): + """流式处理TTS音频,每句只推送一次音频列表""" + payload = { "model": self.model, "text": text, "stream": True, @@ -115,116 +159,183 @@ class TTSProvider(TTSProviderBase): "audio_setting": self.audio_setting, } - if isinstance(self.timber_weights, list) and len(self.timber_weights) > 0: - request_json["timber_weights"] = self.timber_weights - request_json["voice_setting"]["voice_id"] = "" + if type(self.timber_weights) is list and len(self.timber_weights) > 0: + payload["timber_weights"] = self.timber_weights + payload["voice_setting"]["voice_id"] = "" + + frame_bytes = int( + self.opus_encoder.sample_rate + * self.opus_encoder.channels # 1 + * self.opus_encoder.frame_size_ms + / 1000 + * 2 + ) # 16-bit = 2 bytes + try: + async with aiohttp.ClientSession() as session: + async with session.post( + self.api_url, + headers=self.header, + data=json.dumps(payload), + timeout=10, + ) as resp: + + if resp.status != 200: + logger.bind(tag=TAG).error( + f"TTS请求失败: {resp.status}, {await resp.text()}" + ) + self.tts_audio_queue.put((SentenceType.LAST, [], None)) + return + + self.pcm_buffer.clear() + self.tts_audio_queue.put((SentenceType.FIRST, [], text)) + + # 处理音频流数据 + buffer = b"" + async for chunk in resp.content.iter_any(): + if not chunk: + continue + + buffer += chunk + while True: + # 查找数据块分隔符 + header_pos = buffer.find(b"data: ") + if header_pos == -1: + break + + end_pos = buffer.find(b"\n\n", header_pos) + if end_pos == -1: + break + + # 提取单个完整JSON块 + json_str = buffer[header_pos + 6 : end_pos].decode("utf-8") + buffer = buffer[end_pos + 2 :] + + try: + data = json.loads(json_str) + status = data.get("data", {}).get("status", 1) + audio_hex = data.get("data", {}).get("audio") + + # 仅处理status=1的有效音频块 忽略status=2的结束汇总块 + if status == 1 and audio_hex: + pcm_data = bytes.fromhex(audio_hex) + self.pcm_buffer.extend(pcm_data) + + except json.JSONDecodeError as e: + logger.bind(tag=TAG).error(f"JSON解析失败: {e}") + continue + + while len(self.pcm_buffer) >= frame_bytes: + frame = bytes(self.pcm_buffer[:frame_bytes]) + del self.pcm_buffer[:frame_bytes] + + self.opus_encoder.encode_pcm_to_opus_stream( + frame, end_of_stream=False, callback=self.handle_opus + ) + + # flush 剩余不足一帧的数据 + if self.pcm_buffer: + self.opus_encoder.encode_pcm_to_opus_stream( + bytes(self.pcm_buffer), + end_of_stream=True, + callback=self.handle_opus, + ) + self.pcm_buffer.clear() + + # 如果是最后一段,输出音频获取完毕 + if is_last: + self._process_before_stop_play_files() + + except Exception as e: + logger.bind(tag=TAG).error(f"TTS请求异常: {e}") + self.tts_audio_queue.put((SentenceType.LAST, [], None)) + + async def close(self): + """资源清理""" + await super().close() + if hasattr(self, "opus_encoder"): + self.opus_encoder.close() + + def to_tts(self, text: str) -> list: + """非流式TTS处理,用于测试及保存音频文件的场景 + Args: + text: 要转换的文本 + Returns: + list: 返回opus编码后的音频数据列表 + """ + start_time = time.time() + text = MarkdownCleaner.clean_markdown(text) + + payload = { + "model": self.model, + "text": text, + "stream": True, + "voice_setting": self.voice_setting, + "pronunciation_dict": self.pronunciation_dict, + "audio_setting": self.audio_setting, + } + + if type(self.timber_weights) is list and len(self.timber_weights) > 0: + payload["timber_weights"] = self.timber_weights + payload["voice_setting"]["voice_id"] = "" + + headers = { + "Content-Type": "application/json", + "Authorization": f"Bearer {self.api_key}", + } try: with requests.post( - self.api_url, - data=json.dumps(request_json), - headers=self.header, - stream=True + self.api_url, data=json.dumps(payload), headers=headers, timeout=5 ) as response: - - # 检查HTTP状态码 if response.status_code != 200: - raise Exception( - f"HTTP error: {response.status_code}, response: {response.text}" + logger.bind(tag=TAG).error( + f"TTS请求失败: {response.status_code}, {response.text}" ) - - # 处理流式响应 - for line in response.iter_lines(): - if line: # 过滤空行 - # 检查是否为数据行 (SSE格式) - if line.startswith(b'data:'): - try: - data = json.loads(line[5:].strip()) # 去掉"data:"前缀 - - # 检查API状态码 - if data.get("base_resp", {}).get("status_code", -1) != 0: - raise Exception( - f"API error: {data.get('base_resp', {}).get('status_msg')}" - ) - - # 跳过非音频数据块 - if "extra_info" in data: - continue - - # 提取音频数据 - audio_hex = data.get("data", {}).get("audio") - if audio_hex: - audio_chunk = bytes.fromhex(audio_hex) - if chunk_callback: - chunk_callback(audio_chunk) - yield audio_chunk - - except json.JSONDecodeError: - # 忽略JSON解析错误(可能是心跳包等) - continue - except Exception as e: - raise e - - except Exception as e: - raise Exception(f"{__name__} stream error: {e}") + return [] - def save_stream_to_file( - self, - text: str, - output_file: Optional[str] = None, - progress_callback: Optional[callable] = None - ) -> str: - """ - 流式合成并保存到文件 - :param text: 要合成的文本 - :param output_file: 输出文件路径,如果为None则自动生成 - :param progress_callback: 可选的回调函数,接收已写入的字节数 - :return: 保存的文件路径 - """ - if not output_file: - output_file = self.generate_filename(extension=f".{self.audio_file_type}") - - os.makedirs(os.path.dirname(output_file), exist_ok=True) - - total_bytes = 0 - try: - with open(output_file, "wb") as audio_file: - for audio_chunk in self.text_to_speak_stream(text): - audio_file.write(audio_chunk) - audio_file.flush() - total_bytes += len(audio_chunk) - if progress_callback: - progress_callback(total_bytes) - return output_file - except Exception as e: - # 清理可能创建的不完整文件 - if os.path.exists(output_file): - os.remove(output_file) - raise e + logger.info(f"TTS请求成功: {text}, 耗时: {time.time() - start_time}秒") + + # 使用opus编码器处理PCM数据 + opus_datas = [] + full_content = response.content.decode('utf-8') + pcm_data = bytearray() + for data_block in full_content.split('\n\n'): + if not data_block.startswith('data: '): + continue + + try: + json_str = data_block[6:] # 去除'data: '前缀 + data = json.loads(json_str) + if data.get('data', {}).get('status') == 1: + audio_hex = data['data']['audio'] + pcm_data.extend(bytes.fromhex(audio_hex)) + except (json.JSONDecodeError, KeyError) as e: + logger.bind(tag=TAG).warning(f"无效数据块: {e}") + continue + + # 计算每帧的字节数 + frame_bytes = int( + self.opus_encoder.sample_rate + * self.opus_encoder.channels + * self.opus_encoder.frame_size_ms + / 1000 + * 2 + ) + + # 分帧处理合并后的PCM数据 + for i in range(0, len(pcm_data), frame_bytes): + frame = bytes(pcm_data[i:i+frame_bytes]) + if len(frame) < frame_bytes: + frame += b"\x00" * (frame_bytes - len(frame)) + + self.opus_encoder.encode_pcm_to_opus_stream( + frame, + end_of_stream=(i + frame_bytes >= len(pcm_data)), + callback=lambda opus: opus_datas.append(opus) + ) + + return opus_datas - def stream_to_audio_player(self, text: str, player_command: list = None): - """ - 流式合成并直接播放音频 - :param text: 要合成的文本 - :param player_command: 音频播放器命令,默认使用mpv - """ - if player_command is None: - player_command = ["mpv", "--no-cache", "--no-terminal", "--", "fd://0"] - - try: - import subprocess - player_process = subprocess.Popen( - player_command, - stdin=subprocess.PIPE, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - ) - - for audio_chunk in self.text_to_speak_stream(text): - player_process.stdin.write(audio_chunk) - player_process.stdin.flush() - - player_process.stdin.close() - player_process.wait() except Exception as e: - raise Exception(f"Audio player error: {e}") \ No newline at end of file + logger.bind(tag=TAG).error(f"TTS请求异常: {e}") + return [] diff --git a/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py b/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py deleted file mode 100644 index bda6b47f..00000000 --- a/main/xiaozhi-server/core/providers/tts/minimax_webSocket.py +++ /dev/null @@ -1,180 +0,0 @@ -import os -import uuid -import json -import asyncio -import websockets -import ssl -from datetime import datetime -from core.providers.tts.base import TTSProviderBase -from core.utils.util import parse_string_to_list - - -class TTSProvider(TTSProviderBase): - def __init__(self, config, delete_audio_file): - super().__init__(config, delete_audio_file) - self.group_id = config.get("group_id") - self.api_key = config.get("api_key") - self.model = config.get("model") - - # 初始化语音设置 - default_voice_setting = { - "voice_id": "female-shaonv", - "speed": 1, - "vol": 1, - "pitch": 0, - "emotion": "happy", - } - default_pronunciation_dict = {"tone": ["处理/(chu3)(li3)", "危险/dangerous"]} - default_audio_setting = { - "sample_rate": 32000, - "bitrate": 128000, - "format": "mp3", - "channel": 1, - } - - # 合并配置 - self.voice_setting = { - **default_voice_setting, - **config.get("voice_setting", {}), - } - self.pronunciation_dict = { - **default_pronunciation_dict, - **config.get("pronunciation_dict", {}), - } - self.audio_setting = { - **default_audio_setting, - **config.get("audio_setting", {}) - } - self.timber_weights = parse_string_to_list(config.get("timber_weights")) - - # 设置语音ID - if config.get("private_voice"): - self.voice_setting["voice_id"] = config.get("private_voice") - elif config.get("voice_id"): - self.voice_setting["voice_id"] = config.get("voice_id") - - # WebSocket配置 - self.ws_url = "wss://api.minimaxi.com/ws/v1/t2a_v2" - self.headers = { - "Authorization": f"Bearer {self.api_key}", - "GroupId": self.group_id - } - self.audio_file_type = self.audio_setting.get("format", "mp3") - - def generate_filename(self, extension=".mp3"): - """生成唯一的音频文件名""" - return os.path.join( - self.output_file, - f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}", - ) - - async def _establish_connection(self): - """建立WebSocket连接""" - ssl_context = ssl.create_default_context() - ssl_context.check_hostname = False - ssl_context.verify_mode = ssl.CERT_NONE - - try: - ws = await websockets.connect( - self.ws_url, - additional_headers=self.headers, - ssl=ssl_context - ) - connected = json.loads(await ws.recv()) - if connected.get("event") == "connected_success": - print("连接成功") - return ws - return None - except Exception as e: - print(f"连接失败: {e}") - return None - - async def _start_task(self, websocket): - """发送任务开始请求""" - start_msg = { - "event": "task_start", - "model": self.model, - "voice_setting": self.voice_setting, - "pronunciation_dict": self.pronunciation_dict, - "audio_setting": self.audio_setting - } - - if self.timber_weights and len(self.timber_weights) > 0: - start_msg["timber_weights"] = self.timber_weights - start_msg["voice_setting"]["voice_id"] = "" - - await websocket.send(json.dumps(start_msg)) - response = json.loads(await websocket.recv()) - return response.get("event") == "task_started" - - async def _continue_task(self, websocket, text): - """发送继续请求并收集音频数据""" - await websocket.send(json.dumps({ - "event": "task_continue", - "text": text - })) - - audio_chunks = [] - while True: - response = json.loads(await websocket.recv()) - if "data" in response and "audio" in response["data"]: - audio_chunks.append(response["data"]["audio"]) - if response.get("is_final"): - break - return "".join(audio_chunks) - - async def _close_connection(self, websocket): - """关闭连接""" - if websocket: - await websocket.send(json.dumps({"event": "task_finish"})) - await websocket.close() - print("连接已关闭") - - async def text_to_speak(self, text, output_file=None): - """主方法:文本转语音""" - ws = await self._establish_connection() - if not ws: - raise Exception("无法建立WebSocket连接") - - try: - if not await self._start_task(ws): - raise Exception("任务启动失败") - - hex_audio = await self._continue_task(ws, text) - audio_bytes = bytes.fromhex(hex_audio) - - # 保存到文件或返回二进制数据 - if output_file: - with open(output_file, "wb") as f: - f.write(audio_bytes) - print(f"音频已保存为{output_file}") - return output_file - else: - # 返回音频二进制数据(不播放) - return audio_bytes - - finally: - await self._close_connection(ws) - - -async def main(): - """测试用主函数""" - # 示例配置 - config = { - "group_id": "YOUR_GROUP_ID", # 替换为实际的group_id - "api_key": "YOUR_API_KEY", # 替换为实际的api_key - "model": "your-model", # 替换为实际的模型名称 - "voice_id": "male-qn-qingse", - "voice_setting": { - "speed": 1.2, - "emotion": "happy" - } - } - - tts = TTSProvider(config, delete_audio_file=True) - output_file = tts.generate_filename() - await tts.text_to_speak("这是一个测试文本,用于验证流式语音合成功能", output_file) - - -if __name__ == "__main__": - asyncio.run(main()) \ No newline at end of file