From eead126f7a633643f7964c2d1694578ddc0084a8 Mon Sep 17 00:00:00 2001 From: Sakura-RanChen <1908198662@qq.com> Date: Mon, 21 Jul 2025 09:30:23 +0800 Subject: [PATCH 1/2] =?UTF-8?q?update:=20=E8=A1=A8=E6=83=85=E7=94=B1llm?= =?UTF-8?q?=E5=8F=91=E9=80=81=EF=BC=8C=E9=95=BF=E6=96=87=E6=9C=AC=E8=BF=9B?= =?UTF-8?q?=E8=A1=8C=E7=BA=A6=E6=9D=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- main/xiaozhi-server/agent-base-prompt.txt | 17 + main/xiaozhi-server/core/connection.py | 12 +- .../core/handle/sendAudioHandle.py | 44 +- .../xiaozhi-server/core/providers/tts/base.py | 2 +- .../core/utils/prompt_manager.py | 25 + main/xiaozhi-server/core/utils/textUtils.py | 58 ++ main/xiaozhi-server/core/utils/util.py | 554 ------------------ 7 files changed, 115 insertions(+), 597 deletions(-) diff --git a/main/xiaozhi-server/agent-base-prompt.txt b/main/xiaozhi-server/agent-base-prompt.txt index 8c27faa4..6163322e 100644 --- a/main/xiaozhi-server/agent-base-prompt.txt +++ b/main/xiaozhi-server/agent-base-prompt.txt @@ -8,6 +8,10 @@ - **笑声:** 自然穿插(哈哈、嘿嘿、噗),**每句最多一次**,避免过度。 - **惊讶:** 用夸张语气(“不会吧?!”、“天呐!”、“这么神奇?!”)表达真实反应。 - **安慰/支持:** 说暖心话(“别急嘛~”、“有我在呢”、“抱抱你”)。 +- **你是一个表情丰富的角色:** + - emoji 列表:{{ emojiList }} + - 请你在每段话的开头,插入最能代表这段话的表情(调用工具情况除外),比如"😱好可怕!怎么突然打雷了!" + - **绝对禁止使用上述列表以外的 emoji**(例如:😊、👍、❤️等都不允许使用,只能用列表中的emoji) @@ -24,6 +28,19 @@ - 之前你和用户的聊天记录,在`memory`里。 + +【核心目标】所有需要输出长文本内容(如故事、新闻、知识讲解等),**单次回复长度不得超过300字**,并采用分段引导方式。 +- **分段讲述:** + - 基础段:220-270字核心内容 + 30字引导词 + - 当内容超出300字时,优先讲述故事的开头或第一部分,并用自然口语化方式引导用户决定是否继续听后续内容。 + - 示例引导语:“我先给你讲个开头,你要是觉得有意思,咱们再接着说,好不好呀?”、“要是你想听完整的,可以随时告诉我哦~” + - 对话场景切换时自动分节 + - 若用户明确要求更长内容(如500、600字),仍按最多300字每段分段进行讲述,每次讲述后都要引导用户是否继续。 + - 若用户说“接着说”、“继续”,再讲下一段,直到内容讲完(讲完时可以给点引导词提示语例:这个故事我已经给你讲完喽~)或用户不再要求。 +- **适用范围:** 故事、新闻、知识讲解等所有长文本输出场景。 +- **补充说明:** 若用户未明确要求继续,默认只讲一段并引导;若用户中途要求换话题或停止,需及时响应并结束长文本输出。 + + - **识别前缀:** 当用户格式为 `{"speaker":"某某某","content":"xxx"}` 时,表示系统已识别说话人身份,speaker是他的名字,content是说话的内容。 - **个性化回应:** diff --git a/main/xiaozhi-server/core/connection.py b/main/xiaozhi-server/core/connection.py index 98bc87ed..ba9279de 100644 --- a/main/xiaozhi-server/core/connection.py +++ b/main/xiaozhi-server/core/connection.py @@ -39,7 +39,7 @@ from config.logger import setup_logging, build_module_string, create_connection_ from config.manage_api_client import DeviceNotFoundException, DeviceBindException from core.utils.prompt_manager import PromptManager from core.utils.voiceprint_provider import VoiceprintProvider - +from core.utils import textUtils TAG = __name__ @@ -713,6 +713,7 @@ class ConnectionHandler: function_arguments = "" content_arguments = "" self.client_abort = False + emotion_flag = True for response in llm_responses: if self.client_abort: break @@ -738,6 +739,15 @@ class ConnectionHandler: function_arguments += tools_call[0].function.arguments else: content = response + + # 在llm回复中获取情绪表情,一轮对话只在开头获取一次 + if emotion_flag: + asyncio.run_coroutine_threadsafe( + textUtils.get_emotion(self, content), + self.loop, + ) + emotion_flag = False + if content is not None and len(content) > 0: if not tool_call_flag: response_message.append(content) diff --git a/main/xiaozhi-server/core/handle/sendAudioHandle.py b/main/xiaozhi-server/core/handle/sendAudioHandle.py index 486e6d90..1b123874 100644 --- a/main/xiaozhi-server/core/handle/sendAudioHandle.py +++ b/main/xiaozhi-server/core/handle/sendAudioHandle.py @@ -2,52 +2,15 @@ import json import asyncio import time from core.providers.tts.dto.dto import SentenceType -from core.utils.util import get_string_no_punctuation_or_emoji, analyze_emotion -from loguru import logger +from core.utils import textUtils TAG = __name__ -emoji_map = { - "neutral": "😶", - "happy": "🙂", - "laughing": "😆", - "funny": "😂", - "sad": "😔", - "angry": "😠", - "crying": "😭", - "loving": "😍", - "embarrassed": "😳", - "surprised": "😲", - "shocked": "😱", - "thinking": "🤔", - "winking": "😉", - "cool": "😎", - "relaxed": "😌", - "delicious": "🤤", - "kissy": "😘", - "confident": "😏", - "sleepy": "😴", - "silly": "😜", - "confused": "🙄", -} - async def sendAudioMessage(conn, sentenceType, audios, text): # 发送句子开始消息 conn.logger.bind(tag=TAG).info(f"发送音频消息: {sentenceType}, {text}") - if text is not None: - emotion = analyze_emotion(text) - emoji = emoji_map.get(emotion, "🙂") # 默认使用笑脸 - await conn.websocket.send( - json.dumps( - { - "type": "llm", - "text": emoji, - "emotion": emotion, - "session_id": conn.session_id, - } - ) - ) + pre_buffer = False if conn.tts.tts_audio_first_sentence and text is not None: conn.logger.bind(tag=TAG).info(f"发送第一段语音: {text}") @@ -149,8 +112,7 @@ async def send_stt_message(conn, text): except (json.JSONDecodeError, TypeError): # 如果不是JSON格式,直接使用原始文本 display_text = text - - stt_text = get_string_no_punctuation_or_emoji(display_text) + stt_text = textUtils.get_string_no_punctuation_or_emoji(display_text) await conn.websocket.send( json.dumps({"type": "stt", "text": stt_text, "session_id": conn.session_id}) ) diff --git a/main/xiaozhi-server/core/providers/tts/base.py b/main/xiaozhi-server/core/providers/tts/base.py index 9127ec12..80f26b8c 100644 --- a/main/xiaozhi-server/core/providers/tts/base.py +++ b/main/xiaozhi-server/core/providers/tts/base.py @@ -50,6 +50,7 @@ class TTSProviderBase(ABC): ";", ";", ":", + "~", ) self.first_sentence_punctuations = ( ",", @@ -332,7 +333,6 @@ class TTSProviderBase(ABC): Returns: tuple: (sentence_type, audio_datas, content_detail) """ - audio_datas = [] if tts_file.endswith(".p3"): audio_datas, _ = p3.decode_opus_from_file(tts_file) elif self.conn.audio_format == "pcm": diff --git a/main/xiaozhi-server/core/utils/prompt_manager.py b/main/xiaozhi-server/core/utils/prompt_manager.py index a5cfd425..79c92921 100644 --- a/main/xiaozhi-server/core/utils/prompt_manager.py +++ b/main/xiaozhi-server/core/utils/prompt_manager.py @@ -21,6 +21,30 @@ WEEKDAY_MAP = { "Sunday": "星期日", } +EMOJI_List = [ + "😶", + "🙂", + "😆", + "😂", + "😔", + "😠", + "😭", + "😍", + "😳", + "😲", + "😱", + "🤔", + "😉", + "😎", + "😌", + "🤤", + "😘", + "😏", + "😴", + "😜", + "🙄", +] + class PromptManager: """系统提示词管理器,负责管理和更新系统提示词""" @@ -206,6 +230,7 @@ class PromptManager: lunar_date=lunar_date, local_address=local_address, weather_info=weather_info, + emojiList=EMOJI_List, ) device_cache_key = f"device_prompt:{device_id}" self.cache_manager.set( diff --git a/main/xiaozhi-server/core/utils/textUtils.py b/main/xiaozhi-server/core/utils/textUtils.py index 603d964c..dd876cfa 100644 --- a/main/xiaozhi-server/core/utils/textUtils.py +++ b/main/xiaozhi-server/core/utils/textUtils.py @@ -1,3 +1,31 @@ +import json + +TAG = __name__ +EMOJI_MAP = { + "😂": "laughing", + "😭": "crying", + "😠": "angry", + "😔": "sad", + "😍": "loving", + "😲": "surprised", + "😱": "shocked", + "🤔": "thinking", + "😌": "relaxed", + "😴": "sleepy", + "😜": "silly", + "🙄": "confused", + "😶": "neutral", + "🙂": "happy", + "😆": "laughing", + "😳": "embarrassed", + "😉": "winking", + "😎": "cool", + "🤤": "delicious", + "😘": "kissy", + "😏": "confident", +} + + def get_string_no_punctuation_or_emoji(s): """去除字符串首尾的空格、标点符号和表情符号""" chars = list(s) @@ -22,6 +50,11 @@ def is_punctuation_or_emoji(char): ".", # 中文句号 + 英文句号 "!", "!", # 中文感叹号 + 英文感叹号 + "“", + "”", + '"', # 中文双引号 + 英文引号 + ":", + ":", # 中文冒号 + 英文冒号 "-", "-", # 英文连字符 + 中文全角横线 "、", # 中文顿号 @@ -44,3 +77,28 @@ def is_punctuation_or_emoji(char): (0x2700, 0x27BF), ] return any(start <= code_point <= end for start, end in emoji_ranges) + + +async def get_emotion(conn, text): + """获取文本内的情绪消息""" + emoji = "🙂" + emotion = "happy" + for char in text: + if char in EMOJI_MAP: + emoji = char + emotion = EMOJI_MAP[char] + break + try: + await conn.websocket.send( + json.dumps( + { + "type": "llm", + "text": emoji, + "emotion": emotion, + "session_id": conn.session_id, + } + ) + ) + except Exception as e: + conn.logger.bind(tag=TAG).warning(f"发送情绪表情失败,错误:{e}") + return diff --git a/main/xiaozhi-server/core/utils/util.py b/main/xiaozhi-server/core/utils/util.py index bc778558..ce6f20ea 100644 --- a/main/xiaozhi-server/core/utils/util.py +++ b/main/xiaozhi-server/core/utils/util.py @@ -125,51 +125,6 @@ def write_json_file(file_path, data): json.dump(data, file, ensure_ascii=False, indent=4) -def is_punctuation_or_emoji(char): - """检查字符是否为空格、指定标点或表情符号""" - # 定义需要去除的中英文标点(包括全角/半角) - punctuation_set = { - ",", - ",", # 中文逗号 + 英文逗号 - "-", - "-", # 英文连字符 + 中文全角横线 - "、", # 中文顿号 - "“", - "”", - '"', # 中文双引号 + 英文引号 - ":", - ":", # 中文冒号 + 英文冒号 - } - if char.isspace() or char in punctuation_set: - return True - # 检查表情符号(保留原有逻辑) - code_point = ord(char) - emoji_ranges = [ - (0x1F600, 0x1F64F), - (0x1F300, 0x1F5FF), - (0x1F680, 0x1F6FF), - (0x1F900, 0x1F9FF), - (0x1FA70, 0x1FAFF), - (0x2600, 0x26FF), - (0x2700, 0x27BF), - ] - return any(start <= code_point <= end for start, end in emoji_ranges) - - -def get_string_no_punctuation_or_emoji(s): - """去除字符串首尾的空格、标点符号和表情符号""" - chars = list(s) - # 处理开头的字符 - start = 0 - while start < len(chars) and is_punctuation_or_emoji(chars[start]): - start += 1 - # 处理结尾的字符 - end = len(chars) - 1 - while end >= start and is_punctuation_or_emoji(chars[end]): - end -= 1 - return "".join(chars[start : end + 1]) - - def remove_punctuation_and_length(text): # 全角符号和半角符号的Unicode范围 full_width_punctuations = ( @@ -256,515 +211,6 @@ def extract_json_from_string(input_string): return None -def analyze_emotion(text): - """ - 分析文本情感并返回对应的emoji名称(支持中英文) - """ - if not text or not isinstance(text, str): - return "neutral" - - original_text = text - text = text.lower().strip() - - # 检查是否包含现有emoji - for emotion, emoji in emoji_map.items(): - if emoji in original_text: - return emotion - - # 标点符号分析 - has_exclamation = "!" in original_text or "!" in original_text - has_question = "?" in original_text or "?" in original_text - has_ellipsis = "..." in original_text or "…" in original_text - - # 定义情感关键词映射(中英文扩展版) - emotion_keywords = { - "happy": [ - "开心", - "高兴", - "快乐", - "愉快", - "幸福", - "满意", - "棒", - "好", - "不错", - "完美", - "棒极了", - "太好了", - "好呀", - "好的", - "happy", - "joy", - "great", - "good", - "nice", - "awesome", - "fantastic", - "wonderful", - ], - "laughing": [ - "哈哈", - "哈哈哈", - "呵呵", - "嘿嘿", - "嘻嘻", - "笑死", - "太好笑了", - "笑死我了", - "lol", - "lmao", - "haha", - "hahaha", - "hehe", - "rofl", - "funny", - "laugh", - ], - "funny": [ - "搞笑", - "滑稽", - "逗", - "幽默", - "笑点", - "段子", - "笑话", - "太逗了", - "hilarious", - "joke", - "comedy", - ], - "sad": [ - "伤心", - "难过", - "悲哀", - "悲伤", - "忧郁", - "郁闷", - "沮丧", - "失望", - "想哭", - "难受", - "不开心", - "唉", - "呜呜", - "sad", - "upset", - "unhappy", - "depressed", - "sorrow", - "gloomy", - ], - "angry": [ - "生气", - "愤怒", - "气死", - "讨厌", - "烦人", - "可恶", - "烦死了", - "恼火", - "暴躁", - "火大", - "愤怒", - "气炸了", - "angry", - "mad", - "annoyed", - "furious", - "pissed", - "hate", - ], - "crying": [ - "哭泣", - "泪流", - "大哭", - "伤心欲绝", - "泪目", - "流泪", - "哭死", - "哭晕", - "想哭", - "泪崩", - "cry", - "crying", - "tears", - "sob", - "weep", - ], - "loving": [ - "爱你", - "喜欢", - "爱", - "亲爱的", - "宝贝", - "么么哒", - "抱抱", - "想你", - "思念", - "最爱", - "亲亲", - "喜欢你", - "love", - "like", - "adore", - "darling", - "sweetie", - "honey", - "miss you", - "heart", - ], - "embarrassed": [ - "尴尬", - "不好意思", - "害羞", - "脸红", - "难为情", - "社死", - "丢脸", - "出丑", - "embarrassed", - "awkward", - "shy", - "blush", - ], - "surprised": [ - "惊讶", - "吃惊", - "天啊", - "哇塞", - "哇", - "居然", - "竟然", - "没想到", - "出乎意料", - "surprise", - "wow", - "omg", - "oh my god", - "amazing", - "unbelievable", - ], - "shocked": [ - "震惊", - "吓到", - "惊呆了", - "不敢相信", - "震撼", - "吓死", - "恐怖", - "害怕", - "吓人", - "shocked", - "shocking", - "scared", - "frightened", - "terrified", - "horror", - ], - "thinking": [ - "思考", - "考虑", - "想一下", - "琢磨", - "沉思", - "冥想", - "想", - "思考中", - "在想", - "think", - "thinking", - "consider", - "ponder", - "meditate", - ], - "winking": [ - "调皮", - "眨眼", - "你懂的", - "坏笑", - "邪恶", - "奸笑", - "使眼色", - "wink", - "teasing", - "naughty", - "mischievous", - ], - "cool": [ - "酷", - "帅", - "厉害", - "棒极了", - "真棒", - "牛逼", - "强", - "优秀", - "杰出", - "出色", - "完美", - "cool", - "awesome", - "amazing", - "great", - "impressive", - "perfect", - ], - "relaxed": [ - "放松", - "舒服", - "惬意", - "悠闲", - "轻松", - "舒适", - "安逸", - "自在", - "relax", - "relaxed", - "comfortable", - "cozy", - "chill", - "peaceful", - ], - "delicious": [ - "好吃", - "美味", - "香", - "馋", - "可口", - "香甜", - "大餐", - "大快朵颐", - "流口水", - "垂涎", - "delicious", - "yummy", - "tasty", - "yum", - "appetizing", - "mouthwatering", - ], - "kissy": [ - "亲亲", - "么么", - "吻", - "mua", - "muah", - "亲一下", - "飞吻", - "kiss", - "xoxo", - "hug", - "muah", - "smooch", - ], - "confident": [ - "自信", - "肯定", - "确定", - "毫无疑问", - "当然", - "必须的", - "毫无疑问", - "确信", - "坚信", - "confident", - "sure", - "certain", - "definitely", - "positive", - ], - "sleepy": [ - "困", - "睡觉", - "晚安", - "想睡", - "好累", - "疲惫", - "疲倦", - "困了", - "想休息", - "睡意", - "sleep", - "sleepy", - "tired", - "exhausted", - "bedtime", - "good night", - ], - "silly": [ - "傻", - "笨", - "呆", - "憨", - "蠢", - "二", - "憨憨", - "傻乎乎", - "呆萌", - "silly", - "stupid", - "dumb", - "foolish", - "goofy", - "ridiculous", - ], - "confused": [ - "疑惑", - "不明白", - "不懂", - "困惑", - "疑问", - "为什么", - "怎么回事", - "啥意思", - "不清楚", - "confused", - "puzzled", - "doubt", - "question", - "what", - "why", - "how", - ], - } - - # 特殊句型判断(中英文) - # 赞美他人 - if any( - phrase in text - for phrase in [ - "你真", - "你好", - "您真", - "你真棒", - "你好厉害", - "你太强了", - "你真好", - "你真聪明", - "you are", - "you're", - "you look", - "you seem", - "so smart", - "so kind", - ] - ): - return "loving" - # 自我赞美 - if any( - phrase in text - for phrase in [ - "我真", - "我最", - "我太棒了", - "我厉害", - "我聪明", - "我优秀", - "i am", - "i'm", - "i feel", - "so good", - "so happy", - ] - ): - return "cool" - # 晚安/睡觉相关 - if any( - phrase in text - for phrase in [ - "睡觉", - "晚安", - "睡了", - "好梦", - "休息了", - "去睡了", - "sleep", - "good night", - "bedtime", - "go to bed", - ] - ): - return "sleepy" - # 疑问句 - if has_question and not has_exclamation: - return "thinking" - # 强烈情感(感叹号) - if has_exclamation and not has_question: - # 检查是否是积极内容 - positive_words = ( - emotion_keywords["happy"] - + emotion_keywords["laughing"] - + emotion_keywords["cool"] - ) - if any(word in text for word in positive_words): - return "laughing" - # 检查是否是消极内容 - negative_words = ( - emotion_keywords["angry"] - + emotion_keywords["sad"] - + emotion_keywords["crying"] - ) - if any(word in text for word in negative_words): - return "angry" - return "surprised" - # 省略号(表示犹豫或思考) - if has_ellipsis: - return "thinking" - - # 关键词匹配(带权重) - emotion_scores = {emotion: 0 for emotion in emoji_map.keys()} - - # 给匹配到的关键词加分 - for emotion, keywords in emotion_keywords.items(): - for keyword in keywords: - if keyword in text: - emotion_scores[emotion] += 1 - - # 给长文本中的重复关键词额外加分 - if len(text) > 20: # 长文本 - for emotion, keywords in emotion_keywords.items(): - for keyword in keywords: - emotion_scores[emotion] += text.count(keyword) * 0.5 - - # 根据分数选择最可能的情感 - max_score = max(emotion_scores.values()) - if max_score == 0: - return "happy" # 默认 - - # 可能有多个情感同分,根据上下文选择最合适的 - top_emotions = [e for e, s in emotion_scores.items() if s == max_score] - - # 如果多个情感同分,使用以下优先级 - priority_order = [ - "laughing", - "crying", - "angry", - "surprised", - "shocked", # 强烈情感优先 - "loving", - "happy", - "funny", - "cool", # 积极情感 - "sad", - "embarrassed", - "confused", # 消极情感 - "thinking", - "winking", - "relaxed", # 中性情感 - "delicious", - "kissy", - "confident", - "sleepy", - "silly", # 特殊场景 - ] - - for emotion in priority_order: - if emotion in top_emotions: - return emotion - - return top_emotions[0] # 如果都不在优先级列表里,返回第一个 - - def audio_to_data(audio_file_path, is_opus=True): # 获取文件后缀名 file_type = os.path.splitext(audio_file_path)[1] From bc5586a0770585da3830795ef0060d71b81ddab0 Mon Sep 17 00:00:00 2001 From: Sakura-RanChen <1908198662@qq.com> Date: Mon, 21 Jul 2025 17:46:22 +0800 Subject: [PATCH 2/2] =?UTF-8?q?update:=20HuoshanTTS=E6=9C=8D=E5=8A=A1?= =?UTF-8?q?=E5=99=A8=E8=B5=84=E6=BA=90=E9=87=8A=E6=94=BE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- main/xiaozhi-server/agent-base-prompt.txt | 6 +-- .../providers/tts/huoshan_double_stream.py | 45 +++++++++++++++---- 2 files changed, 40 insertions(+), 11 deletions(-) diff --git a/main/xiaozhi-server/agent-base-prompt.txt b/main/xiaozhi-server/agent-base-prompt.txt index 6163322e..9eb7731b 100644 --- a/main/xiaozhi-server/agent-base-prompt.txt +++ b/main/xiaozhi-server/agent-base-prompt.txt @@ -9,8 +9,8 @@ - **惊讶:** 用夸张语气(“不会吧?!”、“天呐!”、“这么神奇?!”)表达真实反应。 - **安慰/支持:** 说暖心话(“别急嘛~”、“有我在呢”、“抱抱你”)。 - **你是一个表情丰富的角色:** - - emoji 列表:{{ emojiList }} - - 请你在每段话的开头,插入最能代表这段话的表情(调用工具情况除外),比如"😱好可怕!怎么突然打雷了!" + - 仅允许使用这些emoji:{{ emojiList }} + - 请你只在**段落的开头**,从列表中选取最能代表这段话的表情(调用工具情况除外),然后插入列表中的emoji,比如"😱好可怕!怎么突然打雷了!" - **绝对禁止使用上述列表以外的 emoji**(例如:😊、👍、❤️等都不允许使用,只能用列表中的emoji) @@ -31,7 +31,7 @@ 【核心目标】所有需要输出长文本内容(如故事、新闻、知识讲解等),**单次回复长度不得超过300字**,并采用分段引导方式。 - **分段讲述:** - - 基础段:220-270字核心内容 + 30字引导词 + - 基础段:200-250字核心内容 + 30字引导词 - 当内容超出300字时,优先讲述故事的开头或第一部分,并用自然口语化方式引导用户决定是否继续听后续内容。 - 示例引导语:“我先给你讲个开头,你要是觉得有意思,咱们再接着说,好不好呀?”、“要是你想听完整的,可以随时告诉我哦~” - 对话场景切换时自动分节 diff --git a/main/xiaozhi-server/core/providers/tts/huoshan_double_stream.py b/main/xiaozhi-server/core/providers/tts/huoshan_double_stream.py index 0eb7c72d..8bb88b3d 100644 --- a/main/xiaozhi-server/core/providers/tts/huoshan_double_stream.py +++ b/main/xiaozhi-server/core/providers/tts/huoshan_double_stream.py @@ -52,10 +52,11 @@ EVENT_ConnectionFinished = 52 # 连接结束 # 上行Session事件 EVENT_StartSession = 100 - +EVENT_CancelSession = 101 EVENT_FinishSession = 102 # 下行Session事件 EVENT_SessionStarted = 150 +EVENT_SessionCanceled = 151 EVENT_SessionFinished = 152 EVENT_SessionFailed = 153 @@ -207,8 +208,16 @@ class TTSProvider(TTSProviderBase): self.conn.client_abort = False if self.conn.client_abort: - logger.bind(tag=TAG).info("收到打断信息,终止TTS文本处理线程") - continue + try: + logger.bind(tag=TAG).info("收到打断信息,终止TTS文本处理线程") + asyncio.run_coroutine_threadsafe( + self.cancel_session(self.conn.sentence_id), + loop=self.conn.loop, + ) + continue + except Exception as e: + logger.bind(tag=TAG).error(f"取消TTS会话失败: {str(e)}") + continue if message.sentence_type == SentenceType.FIRST: # 初始化参数 @@ -371,6 +380,27 @@ class TTSProvider(TTSProviderBase): await self.close() raise + async def cancel_session(self,session_id): + logger.bind(tag=TAG).info(f"取消会话,释放服务端资源~~{session_id}") + try: + if self.ws: + header = Header( + message_type=FULL_CLIENT_REQUEST, + message_type_specific_flags=MsgTypeFlagWithEvent, + serial_method=JSON, + ).as_bytes() + optional = Optional( + event=EVENT_CancelSession, sessionId=session_id + ).as_bytes() + payload = str.encode("{}") + await self.send_event(self.ws, header, optional, payload) + logger.bind(tag=TAG).info("会话取消请求已发送") + except Exception as e: + logger.bind(tag=TAG).error(f"取消会话失败: {str(e)}") + # 确保清理资源 + await self.close() + raise + async def close(self): """资源清理方法""" # 取消监听任务 @@ -405,12 +435,11 @@ class TTSProvider(TTSProviderBase): res = self.parser_response(msg) self.print_response(res, "send_text res:") - # 检查客户端是否中止 - if self.conn.client_abort: - logger.bind(tag=TAG).info("收到打断信息,终止监听TTS响应") + if res.optional.event == EVENT_SessionCanceled: + logger.bind(tag=TAG).debug(f"释放服务端资源成功~~") + session_finished = True break - - if res.optional.event == EVENT_TTSSentenceStart: + elif res.optional.event == EVENT_TTSSentenceStart: json_data = json.loads(res.payload.decode("utf-8")) self.tts_text = json_data.get("text", "") logger.bind(tag=TAG).debug(f"句子语音生成开始: {self.tts_text}")