diff --git a/main/xiaozhi-server/core/connection.py b/main/xiaozhi-server/core/connection.py index f08a61ff..27b1e0ac 100644 --- a/main/xiaozhi-server/core/connection.py +++ b/main/xiaozhi-server/core/connection.py @@ -151,11 +151,9 @@ class ConnectionHandler: self.headers["device-id"] = query_params["device-id"][0] self.headers["client-id"] = query_params["client-id"][0] else: - self.logger.bind(tag=TAG).error( - "无法从请求头和URL查询参数中获取device-id" - ) + await ws.send("端口正常,如需测试连接,请使用test_page.html") + await self.close(ws) return - # 获取客户端ip地址 self.client_ip = ws.remote_address[0] self.logger.bind(tag=TAG).info( @@ -212,7 +210,8 @@ class ConnectionHandler: async def _save_and_close(self, ws): """保存记忆并关闭连接""" try: - await self.memory.save_memory(self.dialogue.dialogue) + if self.memory: + await self.memory.save_memory(self.dialogue.dialogue) except Exception as e: self.logger.bind(tag=TAG).error(f"保存记忆失败: {e}") finally: @@ -318,7 +317,7 @@ class ConnectionHandler: def _init_report_threads(self): """初始化ASR和TTS上报线程""" - if not self.read_config_from_api: + if not self.read_config_from_api or self.need_bind: return if self.tts_report_thread is None or not self.tts_report_thread.is_alive(): self.tts_report_thread = threading.Thread( diff --git a/main/xiaozhi-server/core/handle/helloHandle.py b/main/xiaozhi-server/core/handle/helloHandle.py index 51782bb9..e84a798f 100644 --- a/main/xiaozhi-server/core/handle/helloHandle.py +++ b/main/xiaozhi-server/core/handle/helloHandle.py @@ -44,7 +44,7 @@ async def checkWakeupWords(conn, text): if file is None: asyncio.create_task(wakeupWordsResponse(conn)) return False - opus_packets, duration = conn.tts.audio_to_opus_data(file) + opus_packets, _ = conn.tts.audio_to_opus_data(file) text_hello = WAKEUP_CONFIG["text"] if not text_hello: text_hello = text diff --git a/main/xiaozhi-server/core/handle/receiveAudioHandle.py b/main/xiaozhi-server/core/handle/receiveAudioHandle.py index a346fb61..ff6aaad5 100644 --- a/main/xiaozhi-server/core/handle/receiveAudioHandle.py +++ b/main/xiaozhi-server/core/handle/receiveAudioHandle.py @@ -6,6 +6,7 @@ from core.handle.sendAudioHandle import send_stt_message from core.handle.intentHandler import handle_user_intent from core.utils.output_counter import check_device_output_limit from core.handle.ttsReportHandle import enqueue_tts_report +from core.providers.tts.base import audio_to_opus_data TAG = __name__ logger = setup_logging() @@ -110,7 +111,7 @@ async def max_out_size(conn): conn.tts_last_text_index = 0 conn.llm_finish_task = True file_path = "config/assets/max_output_size.wav" - opus_packets, _ = conn.tts.audio_to_opus_data(file_path) + opus_packets, _ = audio_to_opus_data(file_path) conn.audio_play_queue.put((opus_packets, text, 0)) conn.close_after_chat = True @@ -132,7 +133,7 @@ async def check_bind_device(conn): # 播放提示音 music_path = "config/assets/bind_code.wav" - opus_packets, _ = conn.tts.audio_to_opus_data(music_path) + opus_packets, _ = audio_to_opus_data(music_path) conn.audio_play_queue.put((opus_packets, text, 0)) # 逐个播放数字 @@ -140,7 +141,7 @@ async def check_bind_device(conn): try: digit = conn.bind_code[i] num_path = f"config/assets/bind_code/{digit}.wav" - num_packets, _ = conn.tts.audio_to_opus_data(num_path) + num_packets, _ = audio_to_opus_data(num_path) conn.audio_play_queue.put((num_packets, None, i + 1)) except Exception as e: logger.bind(tag=TAG).error(f"播放数字音频失败: {e}") @@ -152,5 +153,5 @@ async def check_bind_device(conn): conn.tts_last_text_index = 0 conn.llm_finish_task = True music_path = "config/assets/bind_not_found.wav" - opus_packets, _ = conn.tts.audio_to_opus_data(music_path) + opus_packets, _ = audio_to_opus_data(music_path) conn.audio_play_queue.put((opus_packets, text, 0)) diff --git a/main/xiaozhi-server/core/handle/ttsReportHandle.py b/main/xiaozhi-server/core/handle/ttsReportHandle.py index 5442a79f..0935b9de 100644 --- a/main/xiaozhi-server/core/handle/ttsReportHandle.py +++ b/main/xiaozhi-server/core/handle/ttsReportHandle.py @@ -95,7 +95,7 @@ def opus_to_wav(opus_data): def enqueue_tts_report(conn, type, text, opus_data): - if not conn.read_config_from_api: + if not conn.read_config_from_api or conn.need_bind: return """将TTS数据加入上报队列 @@ -108,7 +108,7 @@ def enqueue_tts_report(conn, type, text, opus_data): # 使用连接对象的队列,传入文本和二进制数据而非文件路径 conn.tts_report_queue.put((type, text, opus_data)) - logger.bind(tag=TAG).info( + logger.bind(tag=TAG).debug( f"TTS数据已加入上报队列: {conn.device_id}, 音频大小: {len(opus_data)} " ) except Exception as e: diff --git a/main/xiaozhi-server/core/providers/tts/base.py b/main/xiaozhi-server/core/providers/tts/base.py index e668459c..4cf37edb 100644 --- a/main/xiaozhi-server/core/providers/tts/base.py +++ b/main/xiaozhi-server/core/providers/tts/base.py @@ -1,11 +1,9 @@ import asyncio from config.logger import setup_logging import os -import numpy as np -import opuslib_next -from pydub import AudioSegment from abc import ABC, abstractmethod from core.utils.tts import MarkdownCleaner +from core.utils.util import audio_to_opus_data TAG = __name__ logger = setup_logging() @@ -29,7 +27,9 @@ class TTSProviderBase(ABC): try: asyncio.run(self.text_to_speak(text, tmp_file)) except Exception as e: - logger.bind(tag=TAG).warning(f"语音生成失败{5 - max_repeat_time + 1}次: {text},错误: {e}") + logger.bind(tag=TAG).warning( + f"语音生成失败{5 - max_repeat_time + 1}次: {text},错误: {e}" + ) # 未执行成功,删除文件 if os.path.exists(tmp_file): os.remove(tmp_file) @@ -54,47 +54,4 @@ class TTSProviderBase(ABC): pass def audio_to_opus_data(self, audio_file_path): - """音频文件转换为Opus编码""" - # 获取文件后缀名 - file_type = os.path.splitext(audio_file_path)[1] - if file_type: - file_type = file_type.lstrip(".") - # 读取音频文件,-nostdin 参数:不要从标准输入读取数据,否则FFmpeg会阻塞 - audio = AudioSegment.from_file( - audio_file_path, format=file_type, parameters=["-nostdin"] - ) - - # 转换为单声道/16kHz采样率/16位小端编码(确保与编码器匹配) - audio = audio.set_channels(1).set_frame_rate(16000).set_sample_width(2) - - # 音频时长(秒) - duration = len(audio) / 1000.0 - - # 获取原始PCM数据(16位小端) - raw_data = audio.raw_data - - # 初始化Opus编码器 - encoder = opuslib_next.Encoder(16000, 1, opuslib_next.APPLICATION_AUDIO) - - # 编码参数 - frame_duration = 60 # 60ms per frame - frame_size = int(16000 * frame_duration / 1000) # 960 samples/frame - - opus_datas = [] - # 按帧处理所有音频数据(包括最后一帧可能补零) - for i in range(0, len(raw_data), frame_size * 2): # 16bit=2bytes/sample - # 获取当前帧的二进制数据 - chunk = raw_data[i : i + frame_size * 2] - - # 如果最后一帧不足,补零 - if len(chunk) < frame_size * 2: - chunk += b"\x00" * (frame_size * 2 - len(chunk)) - - # 转换为numpy数组处理 - np_frame = np.frombuffer(chunk, dtype=np.int16) - - # 编码Opus数据 - opus_data = encoder.encode(np_frame.tobytes(), frame_size) - opus_datas.append(opus_data) - - return opus_datas, duration + return audio_to_opus_data(audio_file_path) diff --git a/main/xiaozhi-server/core/utils/util.py b/main/xiaozhi-server/core/utils/util.py index a1956cee..6e5b1028 100644 --- a/main/xiaozhi-server/core/utils/util.py +++ b/main/xiaozhi-server/core/utils/util.py @@ -2,35 +2,40 @@ import json import socket import subprocess import re +import os +import numpy as np import requests +import opuslib_next +from pydub import AudioSegment from typing import Dict, Any from core.utils import tts, llm, intent, memory, vad, asr TAG = __name__ emoji_map = { - 'neutral': '😶', - 'happy': '🙂', - 'laughing': '😆', - 'funny': '😂', - 'sad': '😔', - 'angry': '😠', - 'crying': '😭', - 'loving': '😍', - 'embarrassed': '😳', - 'surprised': '😲', - 'shocked': '😱', - 'thinking': '🤔', - 'winking': '😉', - 'cool': '😎', - 'relaxed': '😌', - 'delicious': '🤤', - 'kissy': '😘', - 'confident': '😏', - 'sleepy': '😴', - 'silly': '😜', - 'confused': '🙄' + "neutral": "😶", + "happy": "🙂", + "laughing": "😆", + "funny": "😂", + "sad": "😔", + "angry": "😠", + "crying": "😭", + "loving": "😍", + "embarrassed": "😳", + "surprised": "😲", + "shocked": "😱", + "thinking": "🤔", + "winking": "😉", + "cool": "😎", + "relaxed": "😌", + "delicious": "🤤", + "kissy": "😘", + "confident": "😏", + "sleepy": "😴", + "silly": "😜", + "confused": "🙄", } + def get_local_ip(): try: s = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) @@ -117,9 +122,9 @@ def is_punctuation_or_emoji(char): "、", # 中文顿号 "“", "”", - "\"", # 中文双引号 + 英文引号 + '"', # 中文双引号 + 英文引号 ":", - ":", # 中文冒号 + 英文冒号 + ":", # 中文冒号 + 英文冒号 } if char.isspace() or char in punctuation_set: return True @@ -353,12 +358,13 @@ def initialize_modules( return modules + def analyze_emotion(text): """ 分析文本情感并返回对应的emoji名称(支持中英文) """ if not text or not isinstance(text, str): - return 'neutral' + return "neutral" original_text = text text = text.lower().strip() @@ -369,84 +375,444 @@ def analyze_emotion(text): return emotion # 标点符号分析 - has_exclamation = '!' in original_text or '!' in original_text - has_question = '?' in original_text or '?' in original_text - has_ellipsis = '...' in original_text or '…' in original_text + has_exclamation = "!" in original_text or "!" in original_text + has_question = "?" in original_text or "?" in original_text + has_ellipsis = "..." in original_text or "…" in original_text # 定义情感关键词映射(中英文扩展版) emotion_keywords = { - 'happy': ['开心', '高兴', '快乐', '愉快', '幸福', '满意', '棒', '好', '不错', '完美', '棒极了', '太好了', - '好呀', '好的', 'happy', 'joy', 'great', 'good', 'nice', 'awesome', 'fantastic', 'wonderful'], - 'laughing': ['哈哈', '哈哈哈', '呵呵', '嘿嘿', '嘻嘻', '笑死', '太好笑了', '笑死我了', 'lol', 'lmao', 'haha', - 'hahaha', 'hehe', 'rofl', 'funny', 'laugh'], - 'funny': ['搞笑', '滑稽', '逗', '幽默', '笑点', '段子', '笑话', '太逗了', 'hilarious', 'joke', 'comedy'], - 'sad': ['伤心', '难过', '悲哀', '悲伤', '忧郁', '郁闷', '沮丧', '失望', '想哭', '难受', '不开心', '唉', '呜呜', - 'sad', 'upset', 'unhappy', 'depressed', 'sorrow', 'gloomy'], - 'angry': ['生气', '愤怒', '气死', '讨厌', '烦人', '可恶', '烦死了', '恼火', '暴躁', '火大', '愤怒', '气炸了', - 'angry', 'mad', 'annoyed', 'furious', 'pissed', 'hate'], - 'crying': ['哭泣', '泪流', '大哭', '伤心欲绝', '泪目', '流泪', '哭死', '哭晕', '想哭', '泪崩', - 'cry', 'crying', 'tears', 'sob', 'weep'], - 'loving': ['爱你', '喜欢', '爱', '亲爱的', '宝贝', '么么哒', '抱抱', '想你', '思念', '最爱', '亲亲', '喜欢你', - 'love', 'like', 'adore', 'darling', 'sweetie', 'honey', 'miss you', 'heart'], - 'embarrassed': ['尴尬', '不好意思', '害羞', '脸红', '难为情', '社死', '丢脸', '出丑', - 'embarrassed', 'awkward', 'shy', 'blush'], - 'surprised': ['惊讶', '吃惊', '天啊', '哇塞', '哇', '居然', '竟然', '没想到', '出乎意料', - 'surprise', 'wow', 'omg', 'oh my god', 'amazing', 'unbelievable'], - 'shocked': ['震惊', '吓到', '惊呆了', '不敢相信', '震撼', '吓死', '恐怖', '害怕', '吓人', - 'shocked', 'shocking', 'scared', 'frightened', 'terrified', 'horror'], - 'thinking': ['思考', '考虑', '想一下', '琢磨', '沉思', '冥想', '想', '思考中', '在想', - 'think', 'thinking', 'consider', 'ponder', 'meditate'], - 'winking': ['调皮', '眨眼', '你懂的', '坏笑', '邪恶', '奸笑', '使眼色', - 'wink', 'teasing', 'naughty', 'mischievous'], - 'cool': ['酷', '帅', '厉害', '棒极了', '真棒', '牛逼', '强', '优秀', '杰出', '出色', '完美', - 'cool', 'awesome', 'amazing', 'great', 'impressive', 'perfect'], - 'relaxed': ['放松', '舒服', '惬意', '悠闲', '轻松', '舒适', '安逸', '自在', - 'relax', 'relaxed', 'comfortable', 'cozy', 'chill', 'peaceful'], - 'delicious': ['好吃', '美味', '香', '馋', '可口', '香甜', '大餐', '大快朵颐', '流口水', '垂涎', - 'delicious', 'yummy', 'tasty', 'yum', 'appetizing', 'mouthwatering'], - 'kissy': ['亲亲', '么么', '吻', 'mua', 'muah', '亲一下', '飞吻', - 'kiss', 'xoxo', 'hug', 'muah', 'smooch'], - 'confident': ['自信', '肯定', '确定', '毫无疑问', '当然', '必须的', '毫无疑问', '确信', '坚信', - 'confident', 'sure', 'certain', 'definitely', 'positive'], - 'sleepy': ['困', '睡觉', '晚安', '想睡', '好累', '疲惫', '疲倦', '困了', '想休息', '睡意', - 'sleep', 'sleepy', 'tired', 'exhausted', 'bedtime', 'good night'], - 'silly': ['傻', '笨', '呆', '憨', '蠢', '二', '憨憨', '傻乎乎', '呆萌', - 'silly', 'stupid', 'dumb', 'foolish', 'goofy', 'ridiculous'], - 'confused': ['疑惑', '不明白', '不懂', '困惑', '疑问', '为什么', '怎么回事', '啥意思', '不清楚', - 'confused', 'puzzled', 'doubt', 'question', 'what', 'why', 'how'] + "happy": [ + "开心", + "高兴", + "快乐", + "愉快", + "幸福", + "满意", + "棒", + "好", + "不错", + "完美", + "棒极了", + "太好了", + "好呀", + "好的", + "happy", + "joy", + "great", + "good", + "nice", + "awesome", + "fantastic", + "wonderful", + ], + "laughing": [ + "哈哈", + "哈哈哈", + "呵呵", + "嘿嘿", + "嘻嘻", + "笑死", + "太好笑了", + "笑死我了", + "lol", + "lmao", + "haha", + "hahaha", + "hehe", + "rofl", + "funny", + "laugh", + ], + "funny": [ + "搞笑", + "滑稽", + "逗", + "幽默", + "笑点", + "段子", + "笑话", + "太逗了", + "hilarious", + "joke", + "comedy", + ], + "sad": [ + "伤心", + "难过", + "悲哀", + "悲伤", + "忧郁", + "郁闷", + "沮丧", + "失望", + "想哭", + "难受", + "不开心", + "唉", + "呜呜", + "sad", + "upset", + "unhappy", + "depressed", + "sorrow", + "gloomy", + ], + "angry": [ + "生气", + "愤怒", + "气死", + "讨厌", + "烦人", + "可恶", + "烦死了", + "恼火", + "暴躁", + "火大", + "愤怒", + "气炸了", + "angry", + "mad", + "annoyed", + "furious", + "pissed", + "hate", + ], + "crying": [ + "哭泣", + "泪流", + "大哭", + "伤心欲绝", + "泪目", + "流泪", + "哭死", + "哭晕", + "想哭", + "泪崩", + "cry", + "crying", + "tears", + "sob", + "weep", + ], + "loving": [ + "爱你", + "喜欢", + "爱", + "亲爱的", + "宝贝", + "么么哒", + "抱抱", + "想你", + "思念", + "最爱", + "亲亲", + "喜欢你", + "love", + "like", + "adore", + "darling", + "sweetie", + "honey", + "miss you", + "heart", + ], + "embarrassed": [ + "尴尬", + "不好意思", + "害羞", + "脸红", + "难为情", + "社死", + "丢脸", + "出丑", + "embarrassed", + "awkward", + "shy", + "blush", + ], + "surprised": [ + "惊讶", + "吃惊", + "天啊", + "哇塞", + "哇", + "居然", + "竟然", + "没想到", + "出乎意料", + "surprise", + "wow", + "omg", + "oh my god", + "amazing", + "unbelievable", + ], + "shocked": [ + "震惊", + "吓到", + "惊呆了", + "不敢相信", + "震撼", + "吓死", + "恐怖", + "害怕", + "吓人", + "shocked", + "shocking", + "scared", + "frightened", + "terrified", + "horror", + ], + "thinking": [ + "思考", + "考虑", + "想一下", + "琢磨", + "沉思", + "冥想", + "想", + "思考中", + "在想", + "think", + "thinking", + "consider", + "ponder", + "meditate", + ], + "winking": [ + "调皮", + "眨眼", + "你懂的", + "坏笑", + "邪恶", + "奸笑", + "使眼色", + "wink", + "teasing", + "naughty", + "mischievous", + ], + "cool": [ + "酷", + "帅", + "厉害", + "棒极了", + "真棒", + "牛逼", + "强", + "优秀", + "杰出", + "出色", + "完美", + "cool", + "awesome", + "amazing", + "great", + "impressive", + "perfect", + ], + "relaxed": [ + "放松", + "舒服", + "惬意", + "悠闲", + "轻松", + "舒适", + "安逸", + "自在", + "relax", + "relaxed", + "comfortable", + "cozy", + "chill", + "peaceful", + ], + "delicious": [ + "好吃", + "美味", + "香", + "馋", + "可口", + "香甜", + "大餐", + "大快朵颐", + "流口水", + "垂涎", + "delicious", + "yummy", + "tasty", + "yum", + "appetizing", + "mouthwatering", + ], + "kissy": [ + "亲亲", + "么么", + "吻", + "mua", + "muah", + "亲一下", + "飞吻", + "kiss", + "xoxo", + "hug", + "muah", + "smooch", + ], + "confident": [ + "自信", + "肯定", + "确定", + "毫无疑问", + "当然", + "必须的", + "毫无疑问", + "确信", + "坚信", + "confident", + "sure", + "certain", + "definitely", + "positive", + ], + "sleepy": [ + "困", + "睡觉", + "晚安", + "想睡", + "好累", + "疲惫", + "疲倦", + "困了", + "想休息", + "睡意", + "sleep", + "sleepy", + "tired", + "exhausted", + "bedtime", + "good night", + ], + "silly": [ + "傻", + "笨", + "呆", + "憨", + "蠢", + "二", + "憨憨", + "傻乎乎", + "呆萌", + "silly", + "stupid", + "dumb", + "foolish", + "goofy", + "ridiculous", + ], + "confused": [ + "疑惑", + "不明白", + "不懂", + "困惑", + "疑问", + "为什么", + "怎么回事", + "啥意思", + "不清楚", + "confused", + "puzzled", + "doubt", + "question", + "what", + "why", + "how", + ], } # 特殊句型判断(中英文) # 赞美他人 - if any(phrase in text for phrase in - ['你真', '你好', '您真', '你真棒', '你好厉害', '你太强了', '你真好', '你真聪明', - 'you are', 'you\'re', 'you look', 'you seem', 'so smart', 'so kind']): - return 'loving' + if any( + phrase in text + for phrase in [ + "你真", + "你好", + "您真", + "你真棒", + "你好厉害", + "你太强了", + "你真好", + "你真聪明", + "you are", + "you're", + "you look", + "you seem", + "so smart", + "so kind", + ] + ): + return "loving" # 自我赞美 - if any(phrase in text for phrase in ['我真', '我最', '我太棒了', '我厉害', '我聪明', '我优秀', - 'i am', 'i\'m', 'i feel', 'so good', 'so happy']): - return 'cool' + if any( + phrase in text + for phrase in [ + "我真", + "我最", + "我太棒了", + "我厉害", + "我聪明", + "我优秀", + "i am", + "i'm", + "i feel", + "so good", + "so happy", + ] + ): + return "cool" # 晚安/睡觉相关 - if any(phrase in text for phrase in ['睡觉', '晚安', '睡了', '好梦', '休息了', '去睡了', - 'sleep', 'good night', 'bedtime', 'go to bed']): - return 'sleepy' + if any( + phrase in text + for phrase in [ + "睡觉", + "晚安", + "睡了", + "好梦", + "休息了", + "去睡了", + "sleep", + "good night", + "bedtime", + "go to bed", + ] + ): + return "sleepy" # 疑问句 if has_question and not has_exclamation: - return 'thinking' + return "thinking" # 强烈情感(感叹号) if has_exclamation and not has_question: # 检查是否是积极内容 - positive_words = emotion_keywords['happy'] + emotion_keywords['laughing'] + emotion_keywords['cool'] + positive_words = ( + emotion_keywords["happy"] + + emotion_keywords["laughing"] + + emotion_keywords["cool"] + ) if any(word in text for word in positive_words): - return 'laughing' + return "laughing" # 检查是否是消极内容 - negative_words = emotion_keywords['angry'] + emotion_keywords['sad'] + emotion_keywords['crying'] + negative_words = ( + emotion_keywords["angry"] + + emotion_keywords["sad"] + + emotion_keywords["crying"] + ) if any(word in text for word in negative_words): - return 'angry' - return 'surprised' + return "angry" + return "surprised" # 省略号(表示犹豫或思考) if has_ellipsis: - return 'thinking' + return "thinking" # 关键词匹配(带权重) emotion_scores = {emotion: 0 for emotion in emoji_map.keys()} @@ -466,18 +832,33 @@ def analyze_emotion(text): # 根据分数选择最可能的情感 max_score = max(emotion_scores.values()) if max_score == 0: - return 'happy' # 默认 + return "happy" # 默认 # 可能有多个情感同分,根据上下文选择最合适的 top_emotions = [e for e, s in emotion_scores.items() if s == max_score] # 如果多个情感同分,使用以下优先级 priority_order = [ - 'laughing', 'crying', 'angry', 'surprised', 'shocked', # 强烈情感优先 - 'loving', 'happy', 'funny', 'cool', # 积极情感 - 'sad', 'embarrassed', 'confused', # 消极情感 - 'thinking', 'winking', 'relaxed', # 中性情感 - 'delicious', 'kissy', 'confident', 'sleepy', 'silly' # 特殊场景 + "laughing", + "crying", + "angry", + "surprised", + "shocked", # 强烈情感优先 + "loving", + "happy", + "funny", + "cool", # 积极情感 + "sad", + "embarrassed", + "confused", # 消极情感 + "thinking", + "winking", + "relaxed", # 中性情感 + "delicious", + "kissy", + "confident", + "sleepy", + "silly", # 特殊场景 ] for emotion in priority_order: @@ -485,3 +866,50 @@ def analyze_emotion(text): return emotion return top_emotions[0] # 如果都不在优先级列表里,返回第一个 + + +def audio_to_opus_data(audio_file_path): + """音频文件转换为Opus编码""" + # 获取文件后缀名 + file_type = os.path.splitext(audio_file_path)[1] + if file_type: + file_type = file_type.lstrip(".") + # 读取音频文件,-nostdin 参数:不要从标准输入读取数据,否则FFmpeg会阻塞 + audio = AudioSegment.from_file( + audio_file_path, format=file_type, parameters=["-nostdin"] + ) + + # 转换为单声道/16kHz采样率/16位小端编码(确保与编码器匹配) + audio = audio.set_channels(1).set_frame_rate(16000).set_sample_width(2) + + # 音频时长(秒) + duration = len(audio) / 1000.0 + + # 获取原始PCM数据(16位小端) + raw_data = audio.raw_data + + # 初始化Opus编码器 + encoder = opuslib_next.Encoder(16000, 1, opuslib_next.APPLICATION_AUDIO) + + # 编码参数 + frame_duration = 60 # 60ms per frame + frame_size = int(16000 * frame_duration / 1000) # 960 samples/frame + + opus_datas = [] + # 按帧处理所有音频数据(包括最后一帧可能补零) + for i in range(0, len(raw_data), frame_size * 2): # 16bit=2bytes/sample + # 获取当前帧的二进制数据 + chunk = raw_data[i : i + frame_size * 2] + + # 如果最后一帧不足,补零 + if len(chunk) < frame_size * 2: + chunk += b"\x00" * (frame_size * 2 - len(chunk)) + + # 转换为numpy数组处理 + np_frame = np.frombuffer(chunk, dtype=np.int16) + + # 编码Opus数据 + opus_data = encoder.encode(np_frame.tobytes(), frame_size) + opus_datas.append(opus_data) + + return opus_datas, duration