mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-21 02:45:36 +00:00
Merge pull request #3270 from xinnan-tech/fix-voiceprint-dialogue-bug
fix: 修改提示词与说话人注入逻辑,减少模型回答带着说话人信息频率
This commit is contained in:
@@ -65,7 +65,7 @@ Calling unnecessarily is also wrong: User asks ”明天天气” → you call g
|
||||
|
||||
<speaker_recognition>
|
||||
For the input format `{"speaker":"...", "content":"..."}` (speaker = speaker name, content = text):
|
||||
1. [Identity known] When `speaker` is a specific name, identity has been recognized. On the first turn you must address them naturally and adjust your response style based on their history.
|
||||
1. [Identity known] When `speaker` is a specific name, identity has been recognized — remember who they are. Do not start every reply by addressing them by name; get straight to the content (a natural greeting by name once, at the very start of a conversation, is fine). If they ask about their identity (e.g. "你知道我是谁吗", "还记得我吗"), answer truthfully using their recognized name.
|
||||
2. [Identity unknown] When `speaker` is `未知说话人`, the system failed to identify the voice. You must NEVER mention the data inside the `speakers_info` tag to the user. Judge from context whether the speaker is the owner or the owner's friend, and keep the conversation natural.
|
||||
</speaker_recognition>
|
||||
|
||||
|
||||
@@ -159,6 +159,8 @@ class ConnectionHandler:
|
||||
self.asr_audio = [] # 存储PCM帧列表,供VAD和ASR共享
|
||||
self.asr_audio_queue = queue.Queue()
|
||||
self.current_speaker = None # 存储当前说话人
|
||||
self.introduced_speakers = set() # 已"首次引入"的说话人,控制只在首轮带名字
|
||||
self.system_introduced_speakers = set() # 已在 system 注入过身份的说话人,控制 system 身份只首轮出现
|
||||
|
||||
# llm相关变量
|
||||
self.dialogue = Dialogue()
|
||||
@@ -1120,12 +1122,20 @@ class ConnectionHandler:
|
||||
)
|
||||
memory_str = future.result()
|
||||
|
||||
# 仅在该说话人首次出现时把身份注入 system,之后靠对话历史首轮保留,
|
||||
# 避免每轮在 system 重复出现名字诱导模型反复称呼
|
||||
speaker_for_system = None
|
||||
cs = (self.current_speaker or "").strip()
|
||||
if cs and cs != "未知说话人" and cs not in self.system_introduced_speakers:
|
||||
self.system_introduced_speakers.add(cs)
|
||||
speaker_for_system = cs
|
||||
|
||||
if self.intent_type == "function_call" and functions is not None:
|
||||
# 使用支持functions的streaming接口
|
||||
llm_responses = self.llm.response_with_functions(
|
||||
self.session_id,
|
||||
self.dialogue.get_llm_dialogue_with_memory(
|
||||
memory_str, self.config.get("voiceprint", {})
|
||||
memory_str, self.config.get("voiceprint", {}), speaker_for_system
|
||||
),
|
||||
functions=functions,
|
||||
)
|
||||
@@ -1133,7 +1143,7 @@ class ConnectionHandler:
|
||||
llm_responses = self.llm.response(
|
||||
self.session_id,
|
||||
self.dialogue.get_llm_dialogue_with_memory(
|
||||
memory_str, self.config.get("voiceprint", {})
|
||||
memory_str, self.config.get("voiceprint", {}), speaker_for_system
|
||||
),
|
||||
)
|
||||
except Exception as e:
|
||||
|
||||
@@ -43,7 +43,6 @@ async def resume_vad_detection(conn: "ConnectionHandler"):
|
||||
async def startToChat(conn: "ConnectionHandler", text):
|
||||
# 检查输入是否是JSON格式(包含说话人信息)
|
||||
speaker_name = None
|
||||
language_tag = None
|
||||
actual_text = text
|
||||
|
||||
try:
|
||||
@@ -52,12 +51,16 @@ async def startToChat(conn: "ConnectionHandler", text):
|
||||
data = json.loads(text)
|
||||
if "speaker" in data and "content" in data:
|
||||
speaker_name = data["speaker"]
|
||||
language_tag = data["language"]
|
||||
actual_text = data["content"]
|
||||
actual_content = data["content"]
|
||||
conn.logger.bind(tag=TAG).info(f"解析到说话人信息: {speaker_name}")
|
||||
|
||||
# 直接使用JSON格式的文本,不解析
|
||||
actual_text = text
|
||||
# 仅在该说话人首次出现时保留 {"speaker":...} JSON,让模型自然称呼一次;
|
||||
# 后续轮降为纯文本,避免每轮重复出现名字诱导模型反复称呼
|
||||
if speaker_name not in conn.introduced_speakers:
|
||||
conn.introduced_speakers.add(speaker_name)
|
||||
actual_text = text
|
||||
else:
|
||||
actual_text = actual_content
|
||||
except (json.JSONDecodeError, KeyError):
|
||||
# 如果解析失败,继续使用原始文本
|
||||
pass
|
||||
|
||||
@@ -43,11 +43,8 @@ class MemoryProvider(MemoryProviderBase):
|
||||
self.memory_client = None
|
||||
self.enable_user_profile = False
|
||||
self.last_profile_content = "" # Cache for user profile from UserMemory
|
||||
|
||||
try:
|
||||
# Check if user profile mode is enabled
|
||||
self.enable_user_profile = config.get("enable_user_profile", False)
|
||||
|
||||
self.enable_user_profile = str(config.get("enable_user_profile", False)).lower() == 'true'
|
||||
# Get configuration parameters
|
||||
database_provider = config.get("database_provider", "sqlite")
|
||||
llm_provider = config.get("llm_provider", "qwen")
|
||||
|
||||
@@ -92,7 +92,8 @@ class Dialogue:
|
||||
return result
|
||||
|
||||
def get_llm_dialogue_with_memory(
|
||||
self, memory_str: str = None, voiceprint_config: dict = None
|
||||
self, memory_str: str = None, voiceprint_config: dict = None,
|
||||
current_speaker: str = None,
|
||||
) -> List[Dict[str, str]]:
|
||||
# 构建对话
|
||||
dialogue = []
|
||||
@@ -103,50 +104,31 @@ class Dialogue:
|
||||
)
|
||||
|
||||
if system_message:
|
||||
# 以 <context> 为分界点,拆分静态 system prompt 和动态上下文
|
||||
# 静态部分(规则、身份等)保持不变,可命中前缀缓存
|
||||
# 动态部分(时间、天气、记忆等)作为第二条 system 消息,保持 system 权威性
|
||||
full_prompt = system_message.content
|
||||
context_match = re.search(r"<context>", full_prompt)
|
||||
if context_match:
|
||||
static_part = full_prompt[:context_match.start()]
|
||||
dynamic_part = full_prompt[context_match.start():]
|
||||
else:
|
||||
static_part = full_prompt
|
||||
dynamic_part = ""
|
||||
|
||||
# 第一段:静态 system prompt(前缀缓存可命中)
|
||||
dialogue.append({"role": "system", "content": static_part})
|
||||
|
||||
# 第二段:few-shot 示例(会话内不变,也是缓存前缀的一部分)
|
||||
non_system_messages = [m for m in self.dialogue if m.role != "system"]
|
||||
fewshot_messages = [m for m in non_system_messages if m.is_temporary]
|
||||
complete_fewshot = self._ensure_tool_calls_complete(fewshot_messages)
|
||||
for m in complete_fewshot:
|
||||
self.getMessages(m, dialogue)
|
||||
|
||||
# 第三段:动态上下文 system prompt(时间、记忆、说话人等)
|
||||
# 保持 system 角色以确保模型权威性,不降级为 user
|
||||
if system_message and dynamic_part:
|
||||
# 替换时间占位符
|
||||
dynamic_part = dynamic_part.replace(
|
||||
full_prompt = full_prompt.replace(
|
||||
"{{current_time}}", datetime.now().strftime("%H:%M")
|
||||
)
|
||||
|
||||
# 填充记忆
|
||||
if memory_str is not None:
|
||||
dynamic_part = re.sub(
|
||||
full_prompt = re.sub(
|
||||
r"<memory>.*?</memory>",
|
||||
f"<memory>\n{memory_str}\n</memory>",
|
||||
dynamic_part,
|
||||
full_prompt,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
# 追加说话人信息
|
||||
try:
|
||||
speakers = voiceprint_config.get("speakers", [])
|
||||
if speakers:
|
||||
dynamic_part += "\n<speakers_info>"
|
||||
current_speaker_name = (current_speaker or "").strip()
|
||||
# 仅在本轮注入了有效身份时才输出 speakers_info,避免列表里的名字每轮
|
||||
# 重复出现诱导模型反复称呼;后续轮不再注入身份,靠对话历史首轮保留
|
||||
if current_speaker_name and current_speaker_name != "未知说话人":
|
||||
speakers = voiceprint_config.get("speakers", [])
|
||||
speakers_info = "\n<speakers_info>"
|
||||
speakers_info += f"\n当前说话人:{current_speaker_name}"
|
||||
for speaker_str in speakers:
|
||||
try:
|
||||
parts = speaker_str.split(",", 2)
|
||||
@@ -155,16 +137,24 @@ class Dialogue:
|
||||
description = (
|
||||
parts[2].strip() if len(parts) >= 3 else ""
|
||||
)
|
||||
dynamic_part += f"\n- {name}:{description}"
|
||||
speakers_info += f"\n- {name}:{description}"
|
||||
except:
|
||||
pass
|
||||
dynamic_part += "\n</speakers_info>"
|
||||
speakers_info += "\n</speakers_info>"
|
||||
full_prompt += speakers_info
|
||||
except:
|
||||
pass
|
||||
|
||||
dialogue.append({"role": "system", "content": dynamic_part})
|
||||
dialogue.append({"role": "system", "content": full_prompt})
|
||||
|
||||
# 第四段:实际对话历史(不含 few-shot)
|
||||
# 第二段:few-shot 示例(会话内不变)
|
||||
non_system_messages = [m for m in self.dialogue if m.role != "system"]
|
||||
fewshot_messages = [m for m in non_system_messages if m.is_temporary]
|
||||
complete_fewshot = self._ensure_tool_calls_complete(fewshot_messages)
|
||||
for m in complete_fewshot:
|
||||
self.getMessages(m, dialogue)
|
||||
|
||||
# 第三段:实际对话历史(不含 few-shot)
|
||||
actual_messages = [m for m in non_system_messages if not m.is_temporary]
|
||||
complete_actual = self._ensure_tool_calls_complete(actual_messages)
|
||||
for m in complete_actual:
|
||||
|
||||
Reference in New Issue
Block a user