1.优化声纹称呼人提示词和注入逻辑,减少次次都称呼的频率
This commit is contained in:
DaGou12138
2026-07-08 16:14:10 +08:00
parent 04c10ef225
commit 3b6e8f0e4b
4 changed files with 26 additions and 13 deletions
+1 -1
View File
@@ -65,7 +65,7 @@ Calling unnecessarily is also wrong: User asks ”明天天气” → you call g
<speaker_recognition> <speaker_recognition>
For the input format `{"speaker":"...", "content":"..."}` (speaker = speaker name, content = text): For the input format `{"speaker":"...", "content":"..."}` (speaker = speaker name, content = text):
1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember it and stay consistent about who they are. The only restriction is about OPENING a reply: do not start every reply by greeting or addressing them by name. You may greet/address them by name naturally ONLY ONCE (the first time you see their name); after that, get straight to the point. This does NOT forbid mentioning their name — if the user asks who they are, whether you know them (e.g. "你知道我是谁吗", "还记得我吗"), or otherwise references their identity, you MUST answer truthfully using their recognized name. Always adjust your response style based on their history. 1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember who they are. Do not start every reply by addressing them by name; get straight to the content (a natural greeting by name once, at the very start of a conversation, is fine). If they ask about their identity (e.g. "你知道我是谁吗", "还记得我吗"), answer truthfully using their recognized name.
2. [Identity unknown] When `speaker` is `未知说话人`, the system failed to identify the voice. You must NEVER mention the data inside the `speakers_info` tag to the user. Judge from context whether the speaker is the owner or the owner's friend, and keep the conversation natural. 2. [Identity unknown] When `speaker` is `未知说话人`, the system failed to identify the voice. You must NEVER mention the data inside the `speakers_info` tag to the user. Judge from context whether the speaker is the owner or the owner's friend, and keep the conversation natural.
</speaker_recognition> </speaker_recognition>
+12 -2
View File
@@ -156,6 +156,8 @@ class ConnectionHandler:
self.asr_audio = [] self.asr_audio = []
self.asr_audio_queue = queue.Queue() self.asr_audio_queue = queue.Queue()
self.current_speaker = None # 存储当前说话人 self.current_speaker = None # 存储当前说话人
self.introduced_speakers = set() # 已"首次引入"的说话人,控制只在首轮带名字
self.system_introduced_speakers = set() # 已在 system 注入过身份的说话人,控制 system 身份只首轮出现
# llm相关变量 # llm相关变量
self.dialogue = Dialogue() self.dialogue = Dialogue()
@@ -978,12 +980,20 @@ class ConnectionHandler:
) )
memory_str = future.result() memory_str = future.result()
# 仅在该说话人首次出现时把身份注入 system,之后靠对话历史首轮保留,
# 避免每轮在 system 重复出现名字诱导模型反复称呼
speaker_for_system = None
cs = (self.current_speaker or "").strip()
if cs and cs != "未知说话人" and cs not in self.system_introduced_speakers:
self.system_introduced_speakers.add(cs)
speaker_for_system = cs
if self.intent_type == "function_call" and functions is not None: if self.intent_type == "function_call" and functions is not None:
# 使用支持functions的streaming接口 # 使用支持functions的streaming接口
llm_responses = self.llm.response_with_functions( llm_responses = self.llm.response_with_functions(
self.session_id, self.session_id,
self.dialogue.get_llm_dialogue_with_memory( self.dialogue.get_llm_dialogue_with_memory(
memory_str, self.config.get("voiceprint", {}), self.current_speaker memory_str, self.config.get("voiceprint", {}), speaker_for_system
), ),
functions=functions, functions=functions,
) )
@@ -991,7 +1001,7 @@ class ConnectionHandler:
llm_responses = self.llm.response( llm_responses = self.llm.response(
self.session_id, self.session_id,
self.dialogue.get_llm_dialogue_with_memory( self.dialogue.get_llm_dialogue_with_memory(
memory_str, self.config.get("voiceprint", {}), self.current_speaker memory_str, self.config.get("voiceprint", {}), speaker_for_system
), ),
) )
except Exception as e: except Exception as e:
@@ -39,7 +39,6 @@ async def resume_vad_detection(conn: "ConnectionHandler"):
async def startToChat(conn: "ConnectionHandler", text): async def startToChat(conn: "ConnectionHandler", text):
# 检查输入是否是JSON格式(包含说话人信息) # 检查输入是否是JSON格式(包含说话人信息)
speaker_name = None speaker_name = None
language_tag = None
actual_text = text actual_text = text
try: try:
@@ -48,12 +47,16 @@ async def startToChat(conn: "ConnectionHandler", text):
data = json.loads(text) data = json.loads(text)
if "speaker" in data and "content" in data: if "speaker" in data and "content" in data:
speaker_name = data["speaker"] speaker_name = data["speaker"]
language_tag = data["language"] actual_content = data["content"]
actual_text = data["content"]
conn.logger.bind(tag=TAG).info(f"解析到说话人信息: {speaker_name}") conn.logger.bind(tag=TAG).info(f"解析到说话人信息: {speaker_name}")
# 直接使用JSON格式的文本,不解析 # 仅在该说话人首次出现时保留 {"speaker":...} JSON,让模型自然称呼一次;
actual_text = text # 后续轮降为纯文本,避免每轮重复出现名字诱导模型反复称呼
if speaker_name not in conn.introduced_speakers:
conn.introduced_speakers.add(speaker_name)
actual_text = text
else:
actual_text = actual_content
except (json.JSONDecodeError, KeyError): except (json.JSONDecodeError, KeyError):
# 如果解析失败,继续使用原始文本 # 如果解析失败,继续使用原始文本
pass pass
+5 -5
View File
@@ -145,13 +145,13 @@ class Dialogue:
# 追加说话人信息 # 追加说话人信息
try: try:
speakers = voiceprint_config.get("speakers", [])
current_speaker_name = (current_speaker or "").strip() current_speaker_name = (current_speaker or "").strip()
if speakers or (current_speaker_name and current_speaker_name != "未知说话人"): # 仅在本轮注入了有效身份时才输出 speakers_info,避免列表里的名字每轮
# 重复出现诱导模型反复称呼;后续轮不再注入身份,靠对话历史首轮保留
if current_speaker_name and current_speaker_name != "未知说话人":
speakers = voiceprint_config.get("speakers", [])
dynamic_part += "\n<speakers_info>" dynamic_part += "\n<speakers_info>"
# 当前说话人置于块首,确保弱模型也能稳定获取身份 dynamic_part += f"\n当前说话人:{current_speaker_name}"
if current_speaker_name and current_speaker_name != "未知说话人":
dynamic_part += f"\n当前说话人:{current_speaker_name}"
for speaker_str in speakers: for speaker_str in speakers:
try: try:
parts = speaker_str.split(",", 2) parts = speaker_str.split(",", 2)