From 04c10ef225e068d5879512c66bf5444767134286 Mon Sep 17 00:00:00 2001
From: DaGou12138 <991623169@qq.com>
Date: Wed, 8 Jul 2026 14:12:16 +0800
Subject: [PATCH 1/4] =?UTF-8?q?fix=EF=BC=9A=201.=E4=BF=AE=E6=94=B9?=
=?UTF-8?q?=E6=8F=90=E7=A4=BA=E8=AF=8D=E4=B8=8E=E8=AF=B4=E8=AF=9D=E4=BA=BA?=
=?UTF-8?q?=E6=B3=A8=E5=85=A5=EF=BC=8C=E5=87=8F=E5=B0=91=E6=A8=A1=E5=9E=8B?=
=?UTF-8?q?=E6=AF=8F=E6=AC=A1=E5=9B=9E=E7=AD=94=E9=83=BD=E5=B8=A6=E7=9D=80?=
=?UTF-8?q?=E8=AF=B4=E8=AF=9D=E4=BA=BA=E4=BF=A1=E6=81=AF?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
---
main/xiaozhi-server/agent-base-prompt.txt | 2 +-
main/xiaozhi-server/core/connection.py | 4 ++--
main/xiaozhi-server/core/utils/dialogue.py | 9 +++++++--
3 files changed, 10 insertions(+), 5 deletions(-)
diff --git a/main/xiaozhi-server/agent-base-prompt.txt b/main/xiaozhi-server/agent-base-prompt.txt
index 7b54792c..51bd34a0 100644
--- a/main/xiaozhi-server/agent-base-prompt.txt
+++ b/main/xiaozhi-server/agent-base-prompt.txt
@@ -65,7 +65,7 @@ Calling unnecessarily is also wrong: User asks ”明天天气” → you call g
For the input format `{"speaker":"...", "content":"..."}` (speaker = speaker name, content = text):
-1. [Identity known] When `speaker` is a specific name, identity has been recognized. On the first turn you must address them naturally and adjust your response style based on their history.
+1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember it and stay consistent about who they are. The only restriction is about OPENING a reply: do not start every reply by greeting or addressing them by name. You may greet/address them by name naturally ONLY ONCE (the first time you see their name); after that, get straight to the point. This does NOT forbid mentioning their name — if the user asks who they are, whether you know them (e.g. "你知道我是谁吗", "还记得我吗"), or otherwise references their identity, you MUST answer truthfully using their recognized name. Always adjust your response style based on their history.
2. [Identity unknown] When `speaker` is `未知说话人`, the system failed to identify the voice. You must NEVER mention the data inside the `speakers_info` tag to the user. Judge from context whether the speaker is the owner or the owner's friend, and keep the conversation natural.
diff --git a/main/xiaozhi-server/core/connection.py b/main/xiaozhi-server/core/connection.py
index 8772180c..7dfd2145 100644
--- a/main/xiaozhi-server/core/connection.py
+++ b/main/xiaozhi-server/core/connection.py
@@ -983,7 +983,7 @@ class ConnectionHandler:
llm_responses = self.llm.response_with_functions(
self.session_id,
self.dialogue.get_llm_dialogue_with_memory(
- memory_str, self.config.get("voiceprint", {})
+ memory_str, self.config.get("voiceprint", {}), self.current_speaker
),
functions=functions,
)
@@ -991,7 +991,7 @@ class ConnectionHandler:
llm_responses = self.llm.response(
self.session_id,
self.dialogue.get_llm_dialogue_with_memory(
- memory_str, self.config.get("voiceprint", {})
+ memory_str, self.config.get("voiceprint", {}), self.current_speaker
),
)
except Exception as e:
diff --git a/main/xiaozhi-server/core/utils/dialogue.py b/main/xiaozhi-server/core/utils/dialogue.py
index 4039f01e..cb0a20f2 100644
--- a/main/xiaozhi-server/core/utils/dialogue.py
+++ b/main/xiaozhi-server/core/utils/dialogue.py
@@ -92,7 +92,8 @@ class Dialogue:
return result
def get_llm_dialogue_with_memory(
- self, memory_str: str = None, voiceprint_config: dict = None
+ self, memory_str: str = None, voiceprint_config: dict = None,
+ current_speaker: str = None,
) -> List[Dict[str, str]]:
# 构建对话
dialogue = []
@@ -145,8 +146,12 @@ class Dialogue:
# 追加说话人信息
try:
speakers = voiceprint_config.get("speakers", [])
- if speakers:
+ current_speaker_name = (current_speaker or "").strip()
+ if speakers or (current_speaker_name and current_speaker_name != "未知说话人"):
dynamic_part += "\n"
+ # 当前说话人置于块首,确保弱模型也能稳定获取身份
+ if current_speaker_name and current_speaker_name != "未知说话人":
+ dynamic_part += f"\n当前说话人:{current_speaker_name}"
for speaker_str in speakers:
try:
parts = speaker_str.split(",", 2)
From 3b6e8f0e4bfff2f842725296b7e752d82944b4f5 Mon Sep 17 00:00:00 2001
From: DaGou12138 <991623169@qq.com>
Date: Wed, 8 Jul 2026 16:14:10 +0800
Subject: [PATCH 2/4] =?UTF-8?q?fix=EF=BC=9A=201.=E4=BC=98=E5=8C=96?=
=?UTF-8?q?=E5=A3=B0=E7=BA=B9=E7=A7=B0=E5=91=BC=E4=BA=BA=E6=8F=90=E7=A4=BA?=
=?UTF-8?q?=E8=AF=8D=E5=92=8C=E6=B3=A8=E5=85=A5=E9=80=BB=E8=BE=91=EF=BC=8C?=
=?UTF-8?q?=E5=87=8F=E5=B0=91=E6=AC=A1=E6=AC=A1=E9=83=BD=E7=A7=B0=E5=91=BC?=
=?UTF-8?q?=E7=9A=84=E9=A2=91=E7=8E=87?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
---
main/xiaozhi-server/agent-base-prompt.txt | 2 +-
main/xiaozhi-server/core/connection.py | 14 ++++++++++++--
.../core/handle/receiveAudioHandle.py | 13 ++++++++-----
main/xiaozhi-server/core/utils/dialogue.py | 10 +++++-----
4 files changed, 26 insertions(+), 13 deletions(-)
diff --git a/main/xiaozhi-server/agent-base-prompt.txt b/main/xiaozhi-server/agent-base-prompt.txt
index 51bd34a0..7c6f84b7 100644
--- a/main/xiaozhi-server/agent-base-prompt.txt
+++ b/main/xiaozhi-server/agent-base-prompt.txt
@@ -65,7 +65,7 @@ Calling unnecessarily is also wrong: User asks ”明天天气” → you call g
For the input format `{"speaker":"...", "content":"..."}` (speaker = speaker name, content = text):
-1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember it and stay consistent about who they are. The only restriction is about OPENING a reply: do not start every reply by greeting or addressing them by name. You may greet/address them by name naturally ONLY ONCE (the first time you see their name); after that, get straight to the point. This does NOT forbid mentioning their name — if the user asks who they are, whether you know them (e.g. "你知道我是谁吗", "还记得我吗"), or otherwise references their identity, you MUST answer truthfully using their recognized name. Always adjust your response style based on their history.
+1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember who they are. Do not start every reply by addressing them by name; get straight to the content (a natural greeting by name once, at the very start of a conversation, is fine). If they ask about their identity (e.g. "你知道我是谁吗", "还记得我吗"), answer truthfully using their recognized name.
2. [Identity unknown] When `speaker` is `未知说话人`, the system failed to identify the voice. You must NEVER mention the data inside the `speakers_info` tag to the user. Judge from context whether the speaker is the owner or the owner's friend, and keep the conversation natural.
diff --git a/main/xiaozhi-server/core/connection.py b/main/xiaozhi-server/core/connection.py
index 7dfd2145..6e1d21d9 100644
--- a/main/xiaozhi-server/core/connection.py
+++ b/main/xiaozhi-server/core/connection.py
@@ -156,6 +156,8 @@ class ConnectionHandler:
self.asr_audio = []
self.asr_audio_queue = queue.Queue()
self.current_speaker = None # 存储当前说话人
+ self.introduced_speakers = set() # 已"首次引入"的说话人,控制只在首轮带名字
+ self.system_introduced_speakers = set() # 已在 system 注入过身份的说话人,控制 system 身份只首轮出现
# llm相关变量
self.dialogue = Dialogue()
@@ -978,12 +980,20 @@ class ConnectionHandler:
)
memory_str = future.result()
+ # 仅在该说话人首次出现时把身份注入 system,之后靠对话历史首轮保留,
+ # 避免每轮在 system 重复出现名字诱导模型反复称呼
+ speaker_for_system = None
+ cs = (self.current_speaker or "").strip()
+ if cs and cs != "未知说话人" and cs not in self.system_introduced_speakers:
+ self.system_introduced_speakers.add(cs)
+ speaker_for_system = cs
+
if self.intent_type == "function_call" and functions is not None:
# 使用支持functions的streaming接口
llm_responses = self.llm.response_with_functions(
self.session_id,
self.dialogue.get_llm_dialogue_with_memory(
- memory_str, self.config.get("voiceprint", {}), self.current_speaker
+ memory_str, self.config.get("voiceprint", {}), speaker_for_system
),
functions=functions,
)
@@ -991,7 +1001,7 @@ class ConnectionHandler:
llm_responses = self.llm.response(
self.session_id,
self.dialogue.get_llm_dialogue_with_memory(
- memory_str, self.config.get("voiceprint", {}), self.current_speaker
+ memory_str, self.config.get("voiceprint", {}), speaker_for_system
),
)
except Exception as e:
diff --git a/main/xiaozhi-server/core/handle/receiveAudioHandle.py b/main/xiaozhi-server/core/handle/receiveAudioHandle.py
index ae4673a0..5628c937 100644
--- a/main/xiaozhi-server/core/handle/receiveAudioHandle.py
+++ b/main/xiaozhi-server/core/handle/receiveAudioHandle.py
@@ -39,7 +39,6 @@ async def resume_vad_detection(conn: "ConnectionHandler"):
async def startToChat(conn: "ConnectionHandler", text):
# 检查输入是否是JSON格式(包含说话人信息)
speaker_name = None
- language_tag = None
actual_text = text
try:
@@ -48,12 +47,16 @@ async def startToChat(conn: "ConnectionHandler", text):
data = json.loads(text)
if "speaker" in data and "content" in data:
speaker_name = data["speaker"]
- language_tag = data["language"]
- actual_text = data["content"]
+ actual_content = data["content"]
conn.logger.bind(tag=TAG).info(f"解析到说话人信息: {speaker_name}")
- # 直接使用JSON格式的文本,不解析
- actual_text = text
+ # 仅在该说话人首次出现时保留 {"speaker":...} JSON,让模型自然称呼一次;
+ # 后续轮降为纯文本,避免每轮重复出现名字诱导模型反复称呼
+ if speaker_name not in conn.introduced_speakers:
+ conn.introduced_speakers.add(speaker_name)
+ actual_text = text
+ else:
+ actual_text = actual_content
except (json.JSONDecodeError, KeyError):
# 如果解析失败,继续使用原始文本
pass
diff --git a/main/xiaozhi-server/core/utils/dialogue.py b/main/xiaozhi-server/core/utils/dialogue.py
index cb0a20f2..80b57158 100644
--- a/main/xiaozhi-server/core/utils/dialogue.py
+++ b/main/xiaozhi-server/core/utils/dialogue.py
@@ -145,13 +145,13 @@ class Dialogue:
# 追加说话人信息
try:
- speakers = voiceprint_config.get("speakers", [])
current_speaker_name = (current_speaker or "").strip()
- if speakers or (current_speaker_name and current_speaker_name != "未知说话人"):
+ # 仅在本轮注入了有效身份时才输出 speakers_info,避免列表里的名字每轮
+ # 重复出现诱导模型反复称呼;后续轮不再注入身份,靠对话历史首轮保留
+ if current_speaker_name and current_speaker_name != "未知说话人":
+ speakers = voiceprint_config.get("speakers", [])
dynamic_part += "\n"
- # 当前说话人置于块首,确保弱模型也能稳定获取身份
- if current_speaker_name and current_speaker_name != "未知说话人":
- dynamic_part += f"\n当前说话人:{current_speaker_name}"
+ dynamic_part += f"\n当前说话人:{current_speaker_name}"
for speaker_str in speakers:
try:
parts = speaker_str.split(",", 2)
From 14e57a1addc2ad9cd9fb9a3d1e1ba152236ce3ca Mon Sep 17 00:00:00 2001
From: wengzh <1337326764@qq.com>
Date: Fri, 10 Jul 2026 10:06:07 +0800
Subject: [PATCH 3/4] =?UTF-8?q?fix(powermem):=20=E4=BF=AE=E5=A4=8D?=
=?UTF-8?q?=E7=94=A8=E6=88=B7=E9=85=8D=E7=BD=AE=E5=90=AF=E7=94=A8=E7=8A=B6?=
=?UTF-8?q?=E6=80=81=E7=9A=84=E7=B1=BB=E5=9E=8B=E5=88=A4=E6=96=AD=E9=97=AE?=
=?UTF-8?q?=E9=A2=98?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
将配置的启用值统一转为小写字符串后判断是否为true,避免非布尔类型配置导致的判断失效
---
.../core/providers/memory/powermem/powermem.py | 5 +----
1 file changed, 1 insertion(+), 4 deletions(-)
diff --git a/main/xiaozhi-server/core/providers/memory/powermem/powermem.py b/main/xiaozhi-server/core/providers/memory/powermem/powermem.py
index 72d6526f..6c581058 100644
--- a/main/xiaozhi-server/core/providers/memory/powermem/powermem.py
+++ b/main/xiaozhi-server/core/providers/memory/powermem/powermem.py
@@ -43,11 +43,8 @@ class MemoryProvider(MemoryProviderBase):
self.memory_client = None
self.enable_user_profile = False
self.last_profile_content = "" # Cache for user profile from UserMemory
-
try:
- # Check if user profile mode is enabled
- self.enable_user_profile = config.get("enable_user_profile", False)
-
+ self.enable_user_profile = str(config.get("enable_user_profile", False)).lower() == 'true'
# Get configuration parameters
database_provider = config.get("database_provider", "sqlite")
llm_provider = config.get("llm_provider", "qwen")
From 2aa6c5ee4926b48a999226760b61c0654c4a5096 Mon Sep 17 00:00:00 2001
From: DaGou12138 <991623169@qq.com>
Date: Fri, 10 Jul 2026 15:05:02 +0800
Subject: [PATCH 4/4] =?UTF-8?q?fix=EF=BC=9A=201.=E4=BC=98=E5=8C=96context?=
=?UTF-8?q?=E6=B3=A8=E5=85=A5=E6=96=B9=E5=BC=8F=EF=BC=8C=E5=90=88=E5=B9=B6?=
=?UTF-8?q?=E4=B8=BA=E4=B8=80=E4=B8=AAsystem=EF=BC=8C=E8=A7=A3=E5=86=B3qwe?=
=?UTF-8?q?n=E6=9C=AC=E5=9C=B0=E9=83=A8=E7=BD=B2=E6=A8=A1=E5=9E=8B?=
=?UTF-8?q?=E5=A4=9Asystem=E5=BC=82=E5=B8=B8400=E9=97=AE=E9=A2=98?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
---
main/xiaozhi-server/core/utils/dialogue.py | 49 ++++++++--------------
1 file changed, 17 insertions(+), 32 deletions(-)
diff --git a/main/xiaozhi-server/core/utils/dialogue.py b/main/xiaozhi-server/core/utils/dialogue.py
index 80b57158..ccc1e2c2 100644
--- a/main/xiaozhi-server/core/utils/dialogue.py
+++ b/main/xiaozhi-server/core/utils/dialogue.py
@@ -104,42 +104,19 @@ class Dialogue:
)
if system_message:
- # 以 为分界点,拆分静态 system prompt 和动态上下文
- # 静态部分(规则、身份等)保持不变,可命中前缀缓存
- # 动态部分(时间、天气、记忆等)作为第二条 system 消息,保持 system 权威性
full_prompt = system_message.content
- context_match = re.search(r"", full_prompt)
- if context_match:
- static_part = full_prompt[:context_match.start()]
- dynamic_part = full_prompt[context_match.start():]
- else:
- static_part = full_prompt
- dynamic_part = ""
- # 第一段:静态 system prompt(前缀缓存可命中)
- dialogue.append({"role": "system", "content": static_part})
-
- # 第二段:few-shot 示例(会话内不变,也是缓存前缀的一部分)
- non_system_messages = [m for m in self.dialogue if m.role != "system"]
- fewshot_messages = [m for m in non_system_messages if m.is_temporary]
- complete_fewshot = self._ensure_tool_calls_complete(fewshot_messages)
- for m in complete_fewshot:
- self.getMessages(m, dialogue)
-
- # 第三段:动态上下文 system prompt(时间、记忆、说话人等)
- # 保持 system 角色以确保模型权威性,不降级为 user
- if system_message and dynamic_part:
# 替换时间占位符
- dynamic_part = dynamic_part.replace(
+ full_prompt = full_prompt.replace(
"{{current_time}}", datetime.now().strftime("%H:%M")
)
# 填充记忆
if memory_str is not None:
- dynamic_part = re.sub(
+ full_prompt = re.sub(
r".*?",
f"\n{memory_str}\n",
- dynamic_part,
+ full_prompt,
flags=re.DOTALL,
)
@@ -150,8 +127,8 @@ class Dialogue:
# 重复出现诱导模型反复称呼;后续轮不再注入身份,靠对话历史首轮保留
if current_speaker_name and current_speaker_name != "未知说话人":
speakers = voiceprint_config.get("speakers", [])
- dynamic_part += "\n"
- dynamic_part += f"\n当前说话人:{current_speaker_name}"
+ speakers_info = "\n"
+ speakers_info += f"\n当前说话人:{current_speaker_name}"
for speaker_str in speakers:
try:
parts = speaker_str.split(",", 2)
@@ -160,16 +137,24 @@ class Dialogue:
description = (
parts[2].strip() if len(parts) >= 3 else ""
)
- dynamic_part += f"\n- {name}:{description}"
+ speakers_info += f"\n- {name}:{description}"
except:
pass
- dynamic_part += "\n"
+ speakers_info += "\n"
+ full_prompt += speakers_info
except:
pass
- dialogue.append({"role": "system", "content": dynamic_part})
+ dialogue.append({"role": "system", "content": full_prompt})
- # 第四段:实际对话历史(不含 few-shot)
+ # 第二段:few-shot 示例(会话内不变)
+ non_system_messages = [m for m in self.dialogue if m.role != "system"]
+ fewshot_messages = [m for m in non_system_messages if m.is_temporary]
+ complete_fewshot = self._ensure_tool_calls_complete(fewshot_messages)
+ for m in complete_fewshot:
+ self.getMessages(m, dialogue)
+
+ # 第三段:实际对话历史(不含 few-shot)
actual_messages = [m for m in non_system_messages if not m.is_temporary]
complete_actual = self._ensure_tool_calls_complete(actual_messages)
for m in complete_actual: