mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-27 17:43:55 +08:00
HuoshanDoubleStreamTTS增加自定义配置
This commit is contained in:
@@ -0,0 +1,87 @@
|
|||||||
|
-- 更新HuoshanDoubleStreamTTS供应器配置,将分散的参数改为JSON字典配置
|
||||||
|
-- 将 speech_rate, loudness_rate, pitch, emotion, emotion_scale 等参数整合为 audio_params, additions, mix_speaker 三个JSON字典
|
||||||
|
|
||||||
|
UPDATE `ai_model_provider`
|
||||||
|
SET `fields` = '[
|
||||||
|
{"key": "ws_url", "type": "string", "label": "WebSocket地址"},
|
||||||
|
{"key": "appid", "type": "string", "label": "应用ID"},
|
||||||
|
{"key": "access_token", "type": "string", "label": "访问令牌"},
|
||||||
|
{"key": "resource_id", "type": "string", "label": "资源ID"},
|
||||||
|
{"key": "speaker", "type": "string", "label": "默认音色"},
|
||||||
|
{"key": "enable_ws_reuse", "type": "boolean", "label": "是否开启链接复用", "default": true},
|
||||||
|
{"key": "audio_params", "type": "dict", "label": "音频输出配置"},
|
||||||
|
{"key": "additions", "type": "dict", "label": "高级文本处理配置"},
|
||||||
|
{"key": "mix_speaker", "type": "dict", "label": "混音控制配置"}
|
||||||
|
]'
|
||||||
|
WHERE `id` = 'SYSTEM_TTS_HSDSTTS';
|
||||||
|
|
||||||
|
-- 更新现有配置,将旧的分散参数迁移到新的JSON字典结构
|
||||||
|
UPDATE `ai_model_config`
|
||||||
|
SET `config_json` = JSON_SET(
|
||||||
|
`config_json`,
|
||||||
|
'$.audio_params', JSON_OBJECT(
|
||||||
|
'speech_rate', CAST(COALESCE(NULLIF(JSON_UNQUOTE(JSON_EXTRACT(`config_json`, '$.speech_rate')), ''), '0') AS SIGNED),
|
||||||
|
'loudness_rate', CAST(COALESCE(NULLIF(JSON_UNQUOTE(JSON_EXTRACT(`config_json`, '$.loudness_rate')), ''), '0') AS SIGNED)
|
||||||
|
),
|
||||||
|
'$.additions', JSON_OBJECT(
|
||||||
|
'aigc_metadata', JSON_OBJECT(),
|
||||||
|
'cache_config', JSON_OBJECT(),
|
||||||
|
'post_process', JSON_OBJECT(
|
||||||
|
'pitch', CAST(COALESCE(NULLIF(JSON_UNQUOTE(JSON_EXTRACT(`config_json`, '$.pitch')), ''), '0') AS SIGNED)
|
||||||
|
)
|
||||||
|
),
|
||||||
|
'$.mix_speaker', JSON_OBJECT()
|
||||||
|
)
|
||||||
|
WHERE `id` = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
|
|
||||||
|
-- 删除旧的分散参数字段
|
||||||
|
UPDATE `ai_model_config`
|
||||||
|
SET `config_json` = JSON_REMOVE(
|
||||||
|
`config_json`,
|
||||||
|
'$.speech_rate',
|
||||||
|
'$.loudness_rate',
|
||||||
|
'$.pitch',
|
||||||
|
'$.emotion',
|
||||||
|
'$.emotion_scale'
|
||||||
|
)
|
||||||
|
WHERE `id` = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
|
|
||||||
|
-- 更新文档链接和备注说明
|
||||||
|
UPDATE `ai_model_config` SET
|
||||||
|
`doc_link` = 'https://www.volcengine.com/docs/6561/1329505',
|
||||||
|
`remark` = '火山引擎双向流式TTS配置说明:
|
||||||
|
1. 访问 https://www.volcengine.com/ 注册并开通火山引擎账号
|
||||||
|
2. 访问 https://console.volcengine.com/speech/service/10007 开通语音合成大模型,购买音色
|
||||||
|
3. 在页面底部获取appid和access_token
|
||||||
|
4. 资源ID固定为:volc.service_type.10029(大模型语音合成及混音)
|
||||||
|
5. 链接复用:开启WebSocket连接复用,默认true减少链接损耗(注意:复用后设备处于聆听状态时空闲链接会占并发数)
|
||||||
|
|
||||||
|
详细参数文档:https://www.volcengine.com/docs/6561/1329505
|
||||||
|
【audio_params】音频输出配置 - 用户可自定义添加火山引擎支持的任何音频参数
|
||||||
|
- speech_rate: 语速(-50~100),默认0
|
||||||
|
- loudness_rate: 音量(-50~100),默认0
|
||||||
|
- emotion: 情感类型(仅部分音色支持),可选值:neutral、happy、sad、angry、fearful、disgusted、surprised
|
||||||
|
- emotion_scale: 情感强度(1~5),默认4
|
||||||
|
示例:{"speech_rate": 10, "loudness_rate": 5, "emotion": "happy", "emotion_scale": 4}
|
||||||
|
|
||||||
|
【additions】高级文本处理配置 - 用户可自定义添加火山引擎支持的任何高级参数
|
||||||
|
- post_process.pitch: 音高(-12~12),默认0
|
||||||
|
- aigc_metadata: AIGC元数据配置
|
||||||
|
- cache_config: 缓存配置
|
||||||
|
示例:{"post_process": {"pitch": 2}, "aigc_metadata": {}, "cache_config": {}}
|
||||||
|
|
||||||
|
【mix_speaker】混音控制配置 - 多音色混合(仅 TTS 1.0)
|
||||||
|
示例:
|
||||||
|
{"speakers": [
|
||||||
|
{"source_speaker": "zh_male_bvlazysheep","mix_factor": 0.3},
|
||||||
|
{"source_speaker": "BV120_streaming","mix_factor": 0.3},
|
||||||
|
{"source_speaker": "zh_male_ahu_conversation_wvae_bigtts","mix_factor": 0.4}
|
||||||
|
]}
|
||||||
|
|
||||||
|
注意:
|
||||||
|
- 多情感音色参数(emotion、emotion_scale)仅部分音色支持
|
||||||
|
- 相关音色列表:https://www.volcengine.com/docs/6561/1257544
|
||||||
|
- 用户可根据火山引擎API文档自行添加更多参数
|
||||||
|
- 混音功能主要适用于豆包语音合成模型1.0的音色,使用时需要将req_params.speaker设置为custom_mix_bigtts
|
||||||
|
'
|
||||||
|
WHERE `id` = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
@@ -494,3 +494,10 @@ databaseChangeLog:
|
|||||||
- sqlFile:
|
- sqlFile:
|
||||||
encoding: utf8
|
encoding: utf8
|
||||||
path: classpath:db/changelog/202601141645.sql
|
path: classpath:db/changelog/202601141645.sql
|
||||||
|
- changeSet:
|
||||||
|
id: 202601231530
|
||||||
|
author: RanChen
|
||||||
|
changes:
|
||||||
|
- sqlFile:
|
||||||
|
encoding: utf8
|
||||||
|
path: classpath:db/changelog/202601231530.sql
|
||||||
|
|||||||
@@ -715,13 +715,31 @@ TTS:
|
|||||||
speaker: zh_female_wanwanxiaohe_moon_bigtts
|
speaker: zh_female_wanwanxiaohe_moon_bigtts
|
||||||
# 开启WebSocket连接复用,默认复用(注意:复用后设备处于聆听状态时空闲链接会占并发数)
|
# 开启WebSocket连接复用,默认复用(注意:复用后设备处于聆听状态时空闲链接会占并发数)
|
||||||
enable_ws_reuse: True
|
enable_ws_reuse: True
|
||||||
speech_rate: 0
|
# 相关参数文档:https://www.volcengine.com/docs/6561/1329505
|
||||||
loudness_rate: 0
|
# 音频输出配置(audio_params)- 用户可自定义添加火山引擎支持的任何音频参数
|
||||||
pitch: 0
|
audio_params:
|
||||||
# 多情感音色参数,注意:当前仅部分音色支持设置情感。
|
speech_rate: 0 # 语速(-50~100)
|
||||||
# 相关音色列表:https://www.volcengine.com/docs/6561/1257544
|
loudness_rate: 0 # 音量(-50~100)
|
||||||
emotion: "neutral" # 情感类型,可选值为:neutral、happy、sad、angry、fearful、disgusted、surprised
|
# 多情感音色参数,注意:当前仅部分音色支持设置情感。
|
||||||
emotion_scale: 4 # 情感强度,可选值为:1~5,默认值为4
|
# 相关音色列表:https://www.volcengine.com/docs/6561/1257544
|
||||||
|
# emotion: "neutral" # 情感类型(仅部分音色支持):neutral、happy、sad、angry、fearful、disgusted、surprised
|
||||||
|
# emotion_scale: 4 # 情感强度(1~5)
|
||||||
|
# 高级文本处理配置(additions)- 用户可自定义添加火山引擎支持的任何高级参数
|
||||||
|
additions:
|
||||||
|
post_process:
|
||||||
|
pitch: 0 # 音高(-12~12)
|
||||||
|
# aigc_metadata: {} # AIGC元数据配置
|
||||||
|
# cache_config: {} # 缓存配置
|
||||||
|
# 混音控制配置(mix_speaker)- 多音色混合(仅 TTS 1.0)
|
||||||
|
# 混音功能主要适用于豆包语音合成模型1.0的音色,使用时需要将req_params.speaker设置为custom_mix_bigtts
|
||||||
|
# mix_speaker:
|
||||||
|
# speakers:
|
||||||
|
# - source_speaker: zh_male_bvlazysheep
|
||||||
|
# mix_factor: 0.3
|
||||||
|
# - source_speaker: BV120_streaming
|
||||||
|
# mix_factor: 0.3
|
||||||
|
# - source_speaker: zh_male_ahu_conversation_wvae_bigtts
|
||||||
|
# mix_factor: 0.4
|
||||||
CosyVoiceSiliconflow:
|
CosyVoiceSiliconflow:
|
||||||
type: siliconflow
|
type: siliconflow
|
||||||
# 硅基流动TTS
|
# 硅基流动TTS
|
||||||
|
|||||||
@@ -154,17 +154,30 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.voice = config.get("private_voice")
|
self.voice = config.get("private_voice")
|
||||||
else:
|
else:
|
||||||
self.voice = config.get("speaker")
|
self.voice = config.get("speaker")
|
||||||
speech_rate = config.get("speech_rate", "0")
|
|
||||||
loudness_rate = config.get("loudness_rate", "0")
|
|
||||||
pitch = config.get("pitch", "0")
|
|
||||||
self.speech_rate = int(speech_rate) if speech_rate else 0
|
|
||||||
self.loudness_rate = int(loudness_rate) if loudness_rate else 0
|
|
||||||
self.pitch = int(pitch) if pitch else 0
|
|
||||||
# 多情感音色参数
|
|
||||||
self.emotion = config.get("emotion", "neutral")
|
|
||||||
emotion_scale = config.get("emotion_scale", "4")
|
|
||||||
self.emotion_scale = int(emotion_scale) if emotion_scale else 4
|
|
||||||
|
|
||||||
|
# 默认 audio_params 配置
|
||||||
|
default_audio_params = {
|
||||||
|
"speech_rate": 0,
|
||||||
|
"loudness_rate": 0
|
||||||
|
}
|
||||||
|
|
||||||
|
# 默认 additions 配置
|
||||||
|
default_additions = {
|
||||||
|
"aigc_metadata": {},
|
||||||
|
"cache_config": {},
|
||||||
|
"post_process": {
|
||||||
|
"pitch": 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# 默认 mix_speaker 配置
|
||||||
|
default_mix_speaker = {}
|
||||||
|
|
||||||
|
# 合并用户配置
|
||||||
|
self.audio_params = {**default_audio_params, **config.get("audio_params", {})}
|
||||||
|
self.additions = {**default_additions, **config.get("additions", {})}
|
||||||
|
self.mix_speaker = {**default_mix_speaker, **config.get("mix_speaker", {})}
|
||||||
|
print(self.mix_speaker)
|
||||||
self.ws_url = config.get("ws_url")
|
self.ws_url = config.get("ws_url")
|
||||||
self.authorization = config.get("authorization")
|
self.authorization = config.get("authorization")
|
||||||
self.header = {"Authorization": f"{self.authorization}{self.access_token}"}
|
self.header = {"Authorization": f"{self.authorization}{self.access_token}"}
|
||||||
@@ -179,6 +192,8 @@ class TTSProvider(TTSProviderBase):
|
|||||||
async def open_audio_channels(self, conn):
|
async def open_audio_channels(self, conn):
|
||||||
try:
|
try:
|
||||||
await super().open_audio_channels(conn)
|
await super().open_audio_channels(conn)
|
||||||
|
# 更新 audio_params 中的采样率为实际的 conn.sample_rate
|
||||||
|
self.audio_params["sample_rate"] = conn.sample_rate
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).error(f"Failed to open audio channels: {str(e)}")
|
logger.bind(tag=TAG).error(f"Failed to open audio channels: {str(e)}")
|
||||||
self.ws = None
|
self.ws = None
|
||||||
@@ -645,18 +660,17 @@ class TTSProvider(TTSProviderBase):
|
|||||||
speaker="",
|
speaker="",
|
||||||
audio_format="pcm",
|
audio_format="pcm",
|
||||||
):
|
):
|
||||||
audio_params = {
|
# 构建 req_params
|
||||||
"format": audio_format,
|
req_params = {
|
||||||
"sample_rate": self.conn.sample_rate,
|
"text": text,
|
||||||
"speech_rate": self.speech_rate,
|
"speaker": speaker,
|
||||||
"loudness_rate": self.loudness_rate
|
"audio_params": {**self.audio_params, "format": audio_format},
|
||||||
|
"additions": json.dumps(self.additions)
|
||||||
}
|
}
|
||||||
|
|
||||||
# 如果是多情感音色,添加情感参数
|
# 如果有 mix_speaker 配置,添加到 req_params
|
||||||
if '_emo_' in self.voice:
|
if self.mix_speaker:
|
||||||
if self.emotion:
|
req_params["mix_speaker"] = self.mix_speaker
|
||||||
audio_params["emotion"] = self.emotion
|
|
||||||
audio_params["emotion_scale"] = self.emotion_scale
|
|
||||||
|
|
||||||
return str.encode(
|
return str.encode(
|
||||||
json.dumps(
|
json.dumps(
|
||||||
@@ -664,17 +678,7 @@ class TTSProvider(TTSProviderBase):
|
|||||||
"user": {"uid": uid},
|
"user": {"uid": uid},
|
||||||
"event": event,
|
"event": event,
|
||||||
"namespace": "BidirectionalTTS",
|
"namespace": "BidirectionalTTS",
|
||||||
"req_params": {
|
"req_params": req_params
|
||||||
"text": text,
|
|
||||||
"speaker": speaker,
|
|
||||||
"audio_params": audio_params,
|
|
||||||
"additions": json.dumps({
|
|
||||||
"post_process": {
|
|
||||||
"pitch": self.pitch
|
|
||||||
}
|
|
||||||
})
|
|
||||||
},
|
|
||||||
|
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user