mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-22 07:03:53 +08:00
Merge branch 'refs/heads/main' into perf-tool-call-optimization
This commit is contained in:
@@ -319,7 +319,7 @@ public interface Constant {
|
||||
/**
|
||||
* 版本号
|
||||
*/
|
||||
public static final String VERSION = "0.9.2";
|
||||
public static final String VERSION = "0.9.3";
|
||||
|
||||
/**
|
||||
* 无效固件URL
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
-- 新增豆包语音合成模型2.0供应器(使用seed-tts-2.0资源ID)
|
||||
-- 与火山双流式TTS配置相同,但resource_id固定为seed-tts-2.0
|
||||
|
||||
-- 插入豆包语音合成模型2.0供应器
|
||||
delete from `ai_model_provider` where id = 'SYSTEM_TTS_HSDSTTS_V2';
|
||||
INSERT INTO `ai_model_provider` (`id`, `model_type`, `provider_code`, `name`, `fields`, `sort`, `creator`, `create_date`, `updater`, `update_date`) VALUES
|
||||
('SYSTEM_TTS_HSDSTTS_V2', 'TTS', 'huoshan_double_stream', '豆包语音合成模型2.0', '[
|
||||
{"key": "ws_url", "type": "string", "label": "WebSocket地址"},
|
||||
{"key": "appid", "type": "string", "label": "应用ID"},
|
||||
{"key": "access_token", "type": "string", "label": "访问令牌"},
|
||||
{"key": "resource_id", "type": "string", "label": "资源ID"},
|
||||
{"key": "speaker", "type": "string", "label": "默认音色"},
|
||||
{"key": "enable_ws_reuse", "type": "boolean", "label": "是否开启链接复用", "default": true},
|
||||
{"key": "audio_params", "type": "dict", "label": "音频输出配置"},
|
||||
{"key": "additions", "type": "dict", "label": "高级文本处理配置"},
|
||||
{"key": "mix_speaker", "type": "dict", "label": "混音控制配置"}
|
||||
]', 14, 1, NOW(), 1, NOW());
|
||||
|
||||
-- 插入豆包语音合成模型2.0配置
|
||||
delete from `ai_model_config` where id = 'TTS_HSDSTTS_V2';
|
||||
INSERT INTO `ai_model_config` VALUES ('TTS_HSDSTTS_V2', 'TTS', 'HuoshanDoubleStreamTTSV2', '豆包语音合成模型2.0', 0, 1, '{
|
||||
"type": "huoshan_double_stream",
|
||||
"ws_url": "wss://openspeech.bytedance.com/api/v3/tts/bidirection",
|
||||
"appid": "",
|
||||
"access_token": "",
|
||||
"resource_id": "seed-tts-2.0",
|
||||
"speaker": "zh_female_xiaohe_uranus_bigtts",
|
||||
"enable_ws_reuse": true,
|
||||
"audio_params": {
|
||||
"speech_rate": 0,
|
||||
"loudness_rate": 0
|
||||
},
|
||||
"additions": {
|
||||
"aigc_metadata": {},
|
||||
"cache_config": {},
|
||||
"post_process": {
|
||||
"pitch": 0
|
||||
}
|
||||
},
|
||||
"mix_speaker": {}
|
||||
}', NULL, NULL, 17, NULL, NULL, NULL, NULL);
|
||||
|
||||
-- 豆包语音合成模型2.0配置说明文档
|
||||
UPDATE `ai_model_config` SET
|
||||
`doc_link` = 'https://www.volcengine.com/docs/6561/1329505',
|
||||
`remark` = '豆包语音合成模型2.0配置说明(基于火山引擎seed-tts-2.0):
|
||||
1. 访问 https://www.volcengine.com/ 注册并开通火山引擎账号
|
||||
2. 访问 https://console.volcengine.com/speech/service/10035 开通语音合成大模型,购买音色
|
||||
3. 在页面底部获取appid和access_token
|
||||
4. 资源ID固定为:seed-tts-2.0(豆包语音合成模型2.0)
|
||||
5. 链接复用:开启WebSocket连接复用,默认true减少链接损耗(注意:复用后设备处于聆听状态时空闲链接会占并发数)
|
||||
|
||||
详细参数文档:https://www.volcengine.com/docs/6561/1329505
|
||||
【audio_params】音频输出配置 - 用户可自定义添加火山引擎支持的任何音频参数
|
||||
- speech_rate: 语速(-50~100),默认0
|
||||
- loudness_rate: 音量(-50~100),默认0
|
||||
示例:{"speech_rate": 10, "loudness_rate": 5}
|
||||
|
||||
【additions】高级文本处理配置 - 用户可自定义添加火山引擎支持的任何高级参数
|
||||
- post_process.pitch: 音高(-12~12),默认0
|
||||
- aigc_metadata: AIGC元数据配置
|
||||
- cache_config: 缓存配置
|
||||
示例:{"post_process": {"pitch": 2}, "aigc_metadata": {}, "cache_config": {}}
|
||||
|
||||
注意:
|
||||
- 豆包语音合成模型2.0使用seed-tts-2.0资源ID,与火山双流式TTS(volc.service_type.10029)不同
|
||||
- 相关音色列表:https://www.volcengine.com/docs/6561/1257544
|
||||
- 用户可根据火山引擎API文档自行添加更多参数
|
||||
' WHERE `id` = 'TTS_HSDSTTS_V2';
|
||||
|
||||
-- 添加豆包语音合成模型2.0音色(与火山双流式TTS音色相同)
|
||||
delete from `ai_tts_voice` where tts_model_id = 'TTS_HSDSTTS_V2';
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0001', 'TTS_HSDSTTS_V2', 'Vivi', 'zh_female_vv_uranus_bigtts', '普通话、日语、印尼语、墨西哥西班牙语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_vv_uranus_bigtts.wav', NULL, NULL, NULL, 1, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0002', 'TTS_HSDSTTS_V2', '小何', 'zh_female_xiaohe_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_xiaohe_uranus_bigtts.mp3', NULL, NULL, NULL, 2, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0003', 'TTS_HSDSTTS_V2', '云舟', 'zh_male_m191_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_m191_uranus_bigtts.mp3', NULL, NULL, NULL, 3, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0004', 'TTS_HSDSTTS_V2', '小天', 'zh_male_taocheng_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_taocheng_uranus_bigtts.mp3', NULL, NULL, NULL, 4, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0005', 'TTS_HSDSTTS_V2', '刘飞', 'zh_male_liufei_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_liufei_uranus_bigtts.mp3', NULL, NULL, NULL, 5, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0006', 'TTS_HSDSTTS_V2', '魅力苏菲', 'zh_female_sophie_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_sophie_uranus_bigtts.mp3', NULL, NULL, NULL, 6, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0007', 'TTS_HSDSTTS_V2', '清新女声', 'zh_female_qingxinnvsheng_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_qingxinnvsheng_uranus_bigtts.mp3', NULL, NULL, NULL, 7, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0008', 'TTS_HSDSTTS_V2', '知性灿灿', 'zh_female_cancan_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_cancan_uranus_bigtts.mp3', NULL, NULL, NULL, 8, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0009', 'TTS_HSDSTTS_V2', '撒娇学妹', 'zh_female_sajiaoxuemei_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_sajiaoxuemei_uranus_bigtts.mp3', NULL, NULL, NULL, 9, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0010', 'TTS_HSDSTTS_V2', '甜美小源', 'zh_female_tianmeixiaoyuan_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_tianmeixiaoyuan_uranus_bigtts.mp3', NULL, NULL, NULL, 10, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0011', 'TTS_HSDSTTS_V2', '甜美桃子', 'zh_female_tianmeitaozi_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_tianmeitaozi_uranus_bigtts.mp3', NULL, NULL, NULL, 11, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0012', 'TTS_HSDSTTS_V2', '爽快思思', 'zh_female_shuangkuaisisi_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_shuangkuaisisi_uranus_bigtts.mp3', NULL, NULL, NULL, 12, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0013', 'TTS_HSDSTTS_V2', '佩奇猪', 'zh_female_peiqi_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_peiqi_uranus_bigtts.mp3', NULL, NULL, NULL, 13, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0014', 'TTS_HSDSTTS_V2', '邻家女孩', 'zh_female_linjianvhai_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_linjianvhai_uranus_bigtts.mp3', NULL, NULL, NULL, 14, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0015', 'TTS_HSDSTTS_V2', '少年梓辛/Brayan', 'zh_male_shaonianzixin_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_shaonianzixin_uranus_bigtts.mp3', NULL, NULL, NULL, 15, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0016', 'TTS_HSDSTTS_V2', '猴哥', 'zh_male_sunwukong_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_sunwukong_uranus_bigtts.mp3', NULL, NULL, NULL, 16, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0017', 'TTS_HSDSTTS_V2', '魅力女友', 'zh_female_meilinvyou_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_meilinvyou_uranus_bigtts.mp3', NULL, NULL, NULL, 17, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0018', 'TTS_HSDSTTS_V2', 'Tim', 'en_male_tim_uranus_bigtts', '英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/en_male_tim_uranus_bigtts.mp3', NULL, NULL, NULL, 18, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0019', 'TTS_HSDSTTS_V2', 'Dacey', 'en_female_dacey_uranus_bigtts', '英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/en_female_dacey_uranus_bigtts.mp3', NULL, NULL, NULL, 19, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0020', 'TTS_HSDSTTS_V2', 'Stokie', 'en_female_stokie_uranus_bigtts', '英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/en_female_stokie_uranus_bigtts.mp3', NULL, NULL, NULL, 20, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0021', 'TTS_HSDSTTS_V2', '温暖阿虎/Alvin', 'zh_male_wennuanahu_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_wennuanahu_uranus_bigtts.mp3', NULL, NULL, NULL, 21, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0022', 'TTS_HSDSTTS_V2', '奶气萌娃', 'zh_male_naiqimengwa_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_naiqimengwa_uranus_bigtts.mp3', NULL, NULL, NULL, 22, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0023', 'TTS_HSDSTTS_V2', '婆婆', 'zh_female_popo_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_popo_uranus_bigtts.mp3', NULL, NULL, NULL, 23, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0024', 'TTS_HSDSTTS_V2', '开朗姐姐', 'zh_female_kailangjiejie_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_kailangjiejie_uranus_bigtts.mp3', NULL, NULL, NULL, 24, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0025', 'TTS_HSDSTTS_V2', '轻盈朵朵', 'saturn_zh_female_qingyingduoduo_cs_tob', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/ICL_zh_female_qingyingduoduo_cs_tob.mp3', NULL, NULL, NULL, 25, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0026', 'TTS_HSDSTTS_V2', '温婉珊珊', 'saturn_zh_female_wenwanshanshan_cs_tob', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/ICL_zh_female_wenwanshanshan_cs_tob.mp3', NULL, NULL, NULL, 26, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0027', 'TTS_HSDSTTS_V2', '霸气青叔', 'zh_male_baqiqingshu_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_baqiqingshu_uranus_bigtts.mp3', NULL, NULL, NULL, 27, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0028', 'TTS_HSDSTTS_V2', '悬疑解说', 'zh_male_xuanyijieshuo_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_xuanyijieshuo_uranus_bigtts.mp3', NULL, NULL, NULL, 28, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0029', 'TTS_HSDSTTS_V2', '古风少御', 'zh_female_gufengshaoyu_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_female_gufengshaoyu_uranus_bigtts.mp3', NULL, NULL, NULL, 29, NULL, NULL, NULL, NULL);
|
||||
INSERT INTO `ai_tts_voice` VALUES ('TTS_HSDSTTS_V2_0030', 'TTS_HSDSTTS_V2', '唐僧', 'zh_male_tangseng_uranus_bigtts', '普通话、英语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/zh_male_tangseng_uranus_bigtts.mp3', NULL, NULL, NULL, 30, NULL, NULL, NULL, NULL);
|
||||
@@ -0,0 +1,45 @@
|
||||
-- 修复豆包语音合成模型2.0 provider_code 重复问题,添加 ASR 2.0 支持
|
||||
|
||||
-- ==================== 豆包语音合成模型2.0 ====================
|
||||
-- 删除 TTS 2.0 供应器(不再需要单独的供应器)
|
||||
delete from `ai_model_provider` where id = 'SYSTEM_TTS_HSDSTTS_V2';
|
||||
|
||||
-- ==================== 豆包语音识别(流式) ====================
|
||||
-- 修正原有豆包语音识别(流式)供应器,移除cluster字段,添加resource_id字段
|
||||
UPDATE `ai_model_provider` SET `fields` = '[{"key":"appid","type":"string","label":"应用ID"},{"key":"access_token","type":"string","label":"访问令牌"},{"key":"boosting_table_name","type":"string","label":"热词文件名称"},{"key":"correct_table_name","type":"string","label":"替换词文件名称"},{"key":"output_dir","type":"string","label":"输出目录"},{"key":"end_window_size","type":"number","label":"静音判定时长(ms)"},{"key":"enable_multilingual","type":"boolean","label":"是否开启多语种识别模式"},{"key":"language","type":"string","label":"指定语言编码"},{"key":"resource_id","type":"string","label":"资源ID"}]' WHERE `id` = 'SYSTEM_ASR_DoubaoStreamASR';
|
||||
|
||||
-- 修正原有豆包语音识别(流式)配置,移除cluster字段,添加resource_id默认值
|
||||
UPDATE `ai_model_config` SET `config_json` = JSON_REMOVE(JSON_SET(`config_json`, '$.resource_id', 'volc.bigasr.sauc.duration'), '$.cluster') WHERE `id` = 'ASR_DoubaoStreamASR';
|
||||
|
||||
-- ==================== 豆包语音识别模型2.0 ====================
|
||||
|
||||
-- 插入豆包语音识别模型2.0配置
|
||||
delete from `ai_model_config` where id = 'ASR_DoubaoStreamASRV2';
|
||||
INSERT INTO `ai_model_config` VALUES ('ASR_DoubaoStreamASRV2', 'ASR', 'DoubaoStreamASRV2', '豆包语音识别模型2.0', 0, 1, '{
|
||||
"type": "doubao_stream",
|
||||
"appid": "",
|
||||
"access_token": "",
|
||||
"resource_id": "volc.seedasr.sauc.duration",
|
||||
"end_window_size": 200,
|
||||
"enable_multilingual": false,
|
||||
"language": "zh-CN",
|
||||
"output_dir": "tmp/"
|
||||
}', NULL, NULL, 6, NULL, NULL, NULL, NULL);
|
||||
|
||||
-- 豆包语音识别模型2.0配置说明文档
|
||||
UPDATE `ai_model_config` SET
|
||||
`doc_link` = 'https://www.volcengine.com/docs/6561/109979',
|
||||
`remark` = '豆包语音识别模型2.0配置说明(基于火山引擎seed-asr):
|
||||
1. 访问 https://www.volcengine.com/ 注册并开通火山引擎账号
|
||||
2. 访问 https://console.volcengine.com/speech/service/10038 开通豆包流式语音识别模型2.0
|
||||
3. 在页面底部获取appid和access_token
|
||||
4. 资源ID有两种:小时版(volc.seedasr.sauc.duration)和并发版(volc.seedasr.sauc.concurrent)
|
||||
- 小时版:固定为:volc.seedasr.sauc.duration(豆包语音识别模型2.0)
|
||||
- 并发版:固定为:volc.seedasr.sauc.concurrent(豆包语音识别模型2.0)
|
||||
|
||||
详细参数文档:https://www.volcengine.com/docs/6561/109979
|
||||
|
||||
注意:
|
||||
- 豆包语音识别模型2.0使用volc.seedasr.sauc.duration资源ID,与豆包语音识别(流式)(volc.bigasr.sauc.duration)不同
|
||||
- 语音识别模型2.0价格更为便宜,建议在高并发场景下使用并发版资源ID
|
||||
' WHERE `id` = 'ASR_DoubaoStreamASRV2';
|
||||
@@ -612,4 +612,18 @@ databaseChangeLog:
|
||||
changes:
|
||||
- sqlFile:
|
||||
encoding: utf8
|
||||
path: classpath:db/changelog/202604161357.sql
|
||||
path: classpath:db/changelog/202604161357.sql
|
||||
- changeSet:
|
||||
id: 202604201719
|
||||
author: RanChen
|
||||
changes:
|
||||
- sqlFile:
|
||||
encoding: utf8
|
||||
path: classpath:db/changelog/202604201719.sql
|
||||
- changeSet:
|
||||
id: 202604211700
|
||||
author: RanChen
|
||||
changes:
|
||||
- sqlFile:
|
||||
encoding: utf8
|
||||
path: classpath:db/changelog/202604211700.sql
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': 'Erneut prüfen',
|
||||
'deviceConfig.connectedXiaozhiHotspot': 'Mit xiaozhi Hotspot verbunden',
|
||||
'deviceConfig.wifiNetwork': 'WiFi-Netzwerk',
|
||||
'deviceConfig.wifiPassword': 'WiFi-Passwort',
|
||||
'deviceConfig.scanning': 'Wird gescannt',
|
||||
'deviceConfig.cancel': 'Abbrechen',
|
||||
'deviceConfig.clickRefreshScan': 'Bitte Scan aktualisieren klicken',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'WiFi-Konfiguration fehlgeschlagen',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': 'Bitte Netzwerkverbindung prüfen',
|
||||
'deviceConfig.startWifiConfigButton': 'Konfiguration starten',
|
||||
'deviceConfig.configuring': 'Wird konfiguriert...',
|
||||
'deviceConfig.wifiConfigInstructions': 'WiFi-Konfigurationsanleitung',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': 'Telefon mit xiaozhi Hotspot verbinden',
|
||||
'deviceConfig.selectTargetWifiNetwork': 'Ziel-WiFi-Netzwerk auswählen',
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': 'Re-check',
|
||||
'deviceConfig.connectedXiaozhiHotspot': 'Connected to xiaozhi hotspot',
|
||||
'deviceConfig.wifiNetwork': 'WiFi Network',
|
||||
'deviceConfig.wifiPassword': 'WiFi Password',
|
||||
'deviceConfig.scanning': 'Scanning',
|
||||
'deviceConfig.cancel': 'Cancel',
|
||||
'deviceConfig.clickRefreshScan': 'Please click Refresh Scan',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'WiFi configuration failed',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': 'Please check network connection',
|
||||
'deviceConfig.startWifiConfigButton': 'Start Configuration',
|
||||
'deviceConfig.configuring': 'Configuring...',
|
||||
'deviceConfig.wifiConfigInstructions': 'WiFi Configuration Instructions',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': 'Phone connect to xiaozhi hotspot',
|
||||
'deviceConfig.selectTargetWifiNetwork': 'Select target WiFi network',
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': 'Verificar Novamente',
|
||||
'deviceConfig.connectedXiaozhiHotspot': 'Conectado ao hotspot xiaozhi',
|
||||
'deviceConfig.wifiNetwork': 'Rede WiFi',
|
||||
'deviceConfig.wifiPassword': 'Senha do WiFi',
|
||||
'deviceConfig.scanning': 'Buscando',
|
||||
'deviceConfig.cancel': 'Cancelar',
|
||||
'deviceConfig.clickRefreshScan': 'Por favor, clique em Atualizar Busca',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'Falha na configuração WiFi',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': 'Por favor, verifique a conexão de rede',
|
||||
'deviceConfig.startWifiConfigButton': 'Iniciar Configuração',
|
||||
'deviceConfig.configuring': 'Configurando...',
|
||||
'deviceConfig.wifiConfigInstructions': 'Instruções de Configuração WiFi',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': 'Conecte o celular ao hotspot xiaozhi',
|
||||
'deviceConfig.selectTargetWifiNetwork': 'Selecione a rede WiFi de destino',
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': 'Kiểm tra lại',
|
||||
'deviceConfig.connectedXiaozhiHotspot': 'Đã kết nối với điểm phát sóng xiaozhi',
|
||||
'deviceConfig.wifiNetwork': 'Mạng WiFi',
|
||||
'deviceConfig.wifiPassword': 'Mật khẩu WiFi',
|
||||
'deviceConfig.scanning': 'Đang quét',
|
||||
'deviceConfig.cancel': 'Hủy',
|
||||
'deviceConfig.clickRefreshScan': 'Vui lòng nhấp Làm mới quét',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'Cấu hình WiFi thất bại',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': 'Vui lòng kiểm tra kết nối mạng',
|
||||
'deviceConfig.startWifiConfigButton': 'Bắt đầu cấu hình',
|
||||
'deviceConfig.configuring': 'Đang cấu hình...',
|
||||
'deviceConfig.wifiConfigInstructions': 'Hướng dẫn cấu hình WiFi',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': 'Điện thoại kết nối với điểm phát sóng xiaozhi',
|
||||
'deviceConfig.selectTargetWifiNetwork': 'Chọn mạng WiFi mục tiêu',
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': '重新检查',
|
||||
'deviceConfig.connectedXiaozhiHotspot': '已连接xiaozhi热点',
|
||||
'deviceConfig.wifiNetwork': 'WiFi网络',
|
||||
'deviceConfig.wifiPassword': 'WiFi密码',
|
||||
'deviceConfig.scanning': '扫描中',
|
||||
'deviceConfig.cancel': '取消',
|
||||
'deviceConfig.clickRefreshScan': '请点击刷新扫描',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'WiFi配网失败',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': '请检查网络连接',
|
||||
'deviceConfig.startWifiConfigButton': '开始配网',
|
||||
'deviceConfig.configuring': '配网中...',
|
||||
'deviceConfig.wifiConfigInstructions': 'WiFi配网说明',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': '手机连接xiaozhi热点',
|
||||
'deviceConfig.selectTargetWifiNetwork': '选择目标WiFi网络',
|
||||
|
||||
@@ -472,6 +472,7 @@ export default {
|
||||
'deviceConfig.reCheck': '重新檢查',
|
||||
'deviceConfig.connectedXiaozhiHotspot': '已連接xiaozhi熱點',
|
||||
'deviceConfig.wifiNetwork': 'WiFi網絡',
|
||||
'deviceConfig.wifiPassword': 'WiFi密碼',
|
||||
'deviceConfig.scanning': '掃描中',
|
||||
'deviceConfig.cancel': '取消',
|
||||
'deviceConfig.clickRefreshScan': '請點擊刷新掃描',
|
||||
@@ -485,6 +486,7 @@ export default {
|
||||
'deviceConfig.wifiConfigFailed': 'WiFi配網失敗',
|
||||
'deviceConfig.pleaseCheckNetworkConnection': '請檢查網絡連接',
|
||||
'deviceConfig.startWifiConfigButton': '開始配網',
|
||||
'deviceConfig.configuring': '配網中...',
|
||||
'deviceConfig.wifiConfigInstructions': 'WiFi配網說明',
|
||||
'deviceConfig.phoneConnectXiaozhiHotspot': '手機連接xiaozhi熱點',
|
||||
'deviceConfig.selectTargetWifiNetwork': '選擇目標WiFi網絡',
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
<script setup lang="ts">
|
||||
import { computed, ref } from 'vue'
|
||||
import { useToast } from 'wot-design-uni'
|
||||
import { t } from '@/i18n'
|
||||
import { toast } from '@/utils/toast'
|
||||
|
||||
// 类型定义
|
||||
interface WiFiNetwork {
|
||||
@@ -19,9 +19,6 @@ interface Props {
|
||||
|
||||
const props = defineProps<Props>()
|
||||
|
||||
// Toast 实例
|
||||
const toast = useToast()
|
||||
|
||||
// 响应式数据
|
||||
const configuring = ref(false)
|
||||
|
||||
@@ -45,7 +42,7 @@ async function checkESP32Connection() {
|
||||
return response.statusCode === 200
|
||||
}
|
||||
catch (error) {
|
||||
console.log(t('deviceConfig.esp32ConnectionCheckFailed') + ':', error)
|
||||
console.log(`${t('deviceConfig.esp32ConnectionCheckFailed')}:`, error)
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -58,12 +55,12 @@ async function submitConfig() {
|
||||
// 检查ESP32连接
|
||||
const connected = await checkESP32Connection()
|
||||
if (!connected) {
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
|
||||
configuring.value = true
|
||||
console.log(t('deviceConfig.startWifiConfig') + ':', props.selectedNetwork.ssid)
|
||||
console.log(`${t('deviceConfig.startWifiConfig')}:`, props.selectedNetwork.ssid)
|
||||
|
||||
try {
|
||||
const response = await uni.request({
|
||||
@@ -83,6 +80,14 @@ async function submitConfig() {
|
||||
|
||||
if (response.statusCode === 200 && (response.data as any)?.success) {
|
||||
toast.success(`${t('deviceConfig.configSuccess')}!${t('deviceConfig.deviceWillConnectTo')} ${props.selectedNetwork.ssid},${t('deviceConfig.deviceWillRestart')}。${t('deviceConfig.pleaseDisconnectXiaozhiHotspot')}`)
|
||||
// 设备退出配网模式
|
||||
setTimeout(() => {
|
||||
uni.request({
|
||||
url: 'http://192.168.4.1/exit',
|
||||
method: 'POST',
|
||||
timeout: 15000,
|
||||
})
|
||||
}, 1500)
|
||||
}
|
||||
else {
|
||||
const errorMsg = (response.data as any)?.error || t('deviceConfig.configFailed')
|
||||
@@ -90,8 +95,8 @@ async function submitConfig() {
|
||||
}
|
||||
}
|
||||
catch (error) {
|
||||
console.error(t('deviceConfig.wifiConfigFailed') + ':', error)
|
||||
toast.error(`${t('deviceConfig.configFailed')},${t('deviceConfig.pleaseCheckNetworkConnection')}`)
|
||||
console.error(`${t('deviceConfig.wifiConfigFailed')}:`, error)
|
||||
toast.error(`${t('deviceConfig.configFailed')},${t('deviceConfig.pleaseCheckNetworkConnection')}`)
|
||||
}
|
||||
finally {
|
||||
configuring.value = false
|
||||
@@ -134,9 +139,9 @@ async function submitConfig() {
|
||||
|
||||
<!-- 使用说明 -->
|
||||
<view class="help-section">
|
||||
<view class="help-title">
|
||||
{{ t('deviceConfig.wifiConfigInstructions') }}
|
||||
</view>
|
||||
<view class="help-title">
|
||||
{{ t('deviceConfig.wifiConfigInstructions') }}
|
||||
</view>
|
||||
<view class="help-content">
|
||||
<text class="help-item">
|
||||
1. {{ t('deviceConfig.phoneConnectXiaozhiHotspot') }} (xiaozhi-XXXXXX)
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
<script setup lang="ts">
|
||||
import { computed, defineEmits, defineExpose, onMounted, ref } from 'vue'
|
||||
import { computed, onMounted, ref } from 'vue'
|
||||
import { useToast } from 'wot-design-uni'
|
||||
import { t } from '@/i18n'
|
||||
|
||||
@@ -71,9 +71,9 @@ async function checkESP32Connection() {
|
||||
// 扫描WiFi网络
|
||||
async function scanWifi() {
|
||||
if (!isConnectedToESP32.value) {
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
|
||||
scanning.value = true
|
||||
console.log('开始扫描WiFi网络')
|
||||
@@ -85,7 +85,7 @@ async function scanWifi() {
|
||||
timeout: 10000,
|
||||
})
|
||||
|
||||
console.log(t('deviceConfig.wifiScanResponse') + ':', response)
|
||||
console.log(`${t('deviceConfig.wifiScanResponse')}:`, response)
|
||||
|
||||
if (response.statusCode === 200 && response.data) {
|
||||
const data = response.data as any
|
||||
@@ -93,6 +93,16 @@ async function scanWifi() {
|
||||
wifiNetworks.value = data.networks
|
||||
console.log(`${t('deviceConfig.scanSuccess')},发现 ${data.networks.length} ${t('deviceConfig.networks')}`)
|
||||
}
|
||||
else if (data.aps && Array.isArray(data.aps)) {
|
||||
// 兼容 { data: { support_5g, aps } } 格式
|
||||
wifiNetworks.value = data.aps.map((item: any) => ({
|
||||
ssid: item.ssid,
|
||||
rssi: item.rssi,
|
||||
authmode: item.authmode,
|
||||
channel: item.channel || 0,
|
||||
}))
|
||||
console.log(`${t('deviceConfig.scanSuccess')},发现 ${data.aps.length} ${t('deviceConfig.networks')}`)
|
||||
}
|
||||
else if (Array.isArray(response.data)) {
|
||||
// 兼容旧格式
|
||||
wifiNetworks.value = response.data.map((item: any) => ({
|
||||
@@ -111,8 +121,8 @@ async function scanWifi() {
|
||||
}
|
||||
}
|
||||
catch (error) {
|
||||
console.error(t('deviceConfig.wifiScanFailed') + ':', error)
|
||||
toast.error(t('deviceConfig.scanFailedCheckConnection'))
|
||||
console.error(`${t('deviceConfig.wifiScanFailed')}:`, error)
|
||||
toast.error(t('deviceConfig.scanFailedCheckConnection'))
|
||||
}
|
||||
finally {
|
||||
scanning.value = false
|
||||
@@ -125,9 +135,9 @@ async function showNetworkSelector() {
|
||||
await checkESP32Connection()
|
||||
|
||||
if (!isConnectedToESP32.value) {
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
toast.error(t('deviceConfig.connectXiaozhiHotspot'))
|
||||
return
|
||||
}
|
||||
|
||||
selectorExpanded.value = true
|
||||
|
||||
@@ -213,73 +223,73 @@ onMounted(() => {
|
||||
<!-- Xiaozhi连接状态 -->
|
||||
<view v-if="props.autoConnect" class="connection-status">
|
||||
<view v-if="!isConnectedToESP32" class="status-warning">
|
||||
<view class="status-content">
|
||||
<text class="warning-text">
|
||||
{{ t('deviceConfig.connectXiaozhiHotspot') }} (xiaozhi-XXXXXX)
|
||||
</text>
|
||||
<wd-button
|
||||
size="small"
|
||||
type="primary"
|
||||
:loading="checkingConnection"
|
||||
@click="checkESP32Connection"
|
||||
>
|
||||
{{ checkingConnection ? t('deviceConfig.checking') : t('deviceConfig.reCheck') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
<view class="status-content">
|
||||
<text class="warning-text">
|
||||
{{ t('deviceConfig.connectXiaozhiHotspot') }} (xiaozhi-XXXXXX)
|
||||
</text>
|
||||
<wd-button
|
||||
size="small"
|
||||
type="primary"
|
||||
:loading="checkingConnection"
|
||||
@click="checkESP32Connection"
|
||||
>
|
||||
{{ checkingConnection ? t('deviceConfig.checking') : t('deviceConfig.reCheck') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
</view>
|
||||
<view v-else class="status-success">
|
||||
<view class="status-content">
|
||||
<text class="success-text">
|
||||
{{ t('deviceConfig.connectedXiaozhiHotspot') }}
|
||||
</text>
|
||||
<wd-button
|
||||
size="small"
|
||||
:loading="checkingConnection"
|
||||
@click="checkESP32Connection"
|
||||
>
|
||||
{{ checkingConnection ? t('deviceConfig.checking') : t('deviceConfig.refreshStatus') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
<view class="status-content">
|
||||
<text class="success-text">
|
||||
{{ t('deviceConfig.connectedXiaozhiHotspot') }}
|
||||
</text>
|
||||
<wd-button
|
||||
size="small"
|
||||
:loading="checkingConnection"
|
||||
@click="checkESP32Connection"
|
||||
>
|
||||
{{ checkingConnection ? t('deviceConfig.checking') : t('deviceConfig.refreshStatus') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
</view>
|
||||
</view>
|
||||
|
||||
<!-- WiFi网络选择器 -->
|
||||
<view class="network-selector">
|
||||
<view class="selector-item" @click="showNetworkSelector">
|
||||
<text class="selector-label">
|
||||
{{ t('deviceConfig.wifiNetwork') }}
|
||||
</text>
|
||||
<text class="selector-value">
|
||||
{{ networkDisplayText }}
|
||||
</text>
|
||||
<wd-icon name="arrow-right" custom-class="arrow-icon" />
|
||||
</view>
|
||||
<view class="selector-item" @click="showNetworkSelector">
|
||||
<text class="selector-label">
|
||||
{{ t('deviceConfig.wifiNetwork') }}
|
||||
</text>
|
||||
<text class="selector-value">
|
||||
{{ networkDisplayText }}
|
||||
</text>
|
||||
<wd-icon name="arrow-right" custom-class="arrow-icon" />
|
||||
</view>
|
||||
</view>
|
||||
|
||||
<!-- 展开的网络列表 -->
|
||||
<view v-if="selectorExpanded" class="network-list-overlay">
|
||||
<view class="network-list-container">
|
||||
<view class="list-header">
|
||||
<text class="list-title">
|
||||
{{ t('deviceConfig.selectWifiNetwork') }}
|
||||
</text>
|
||||
<view class="list-actions">
|
||||
<wd-button
|
||||
type="primary"
|
||||
size="small"
|
||||
:loading="scanning"
|
||||
@click="scanWifi"
|
||||
>
|
||||
{{ scanning ? t('deviceConfig.scanning') : t('deviceConfig.refreshScan') }}
|
||||
</wd-button>
|
||||
<wd-button
|
||||
size="small"
|
||||
@click="selectorExpanded = false"
|
||||
>
|
||||
{{ t('deviceConfig.cancel') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
<text class="list-title">
|
||||
{{ t('deviceConfig.selectWifiNetwork') }}
|
||||
</text>
|
||||
<view class="list-actions">
|
||||
<wd-button
|
||||
type="primary"
|
||||
size="small"
|
||||
:loading="scanning"
|
||||
@click="scanWifi"
|
||||
>
|
||||
{{ scanning ? t('deviceConfig.scanning') : t('deviceConfig.refreshScan') }}
|
||||
</wd-button>
|
||||
<wd-button
|
||||
size="small"
|
||||
@click="selectorExpanded = false"
|
||||
>
|
||||
{{ t('deviceConfig.cancel') }}
|
||||
</wd-button>
|
||||
</view>
|
||||
</view>
|
||||
|
||||
<view class="network-list">
|
||||
<view v-if="wifiNetworks.length === 0 && !scanning" class="empty-state">
|
||||
@@ -313,7 +323,7 @@ onMounted(() => {
|
||||
</view>
|
||||
<view class="wifi-security">
|
||||
<text class="security-icon">
|
||||
{{ network.authmode === 0 ? t('deviceConfig.open') : t('deviceConfig.encrypted') }}
|
||||
{{ network.authmode === 0 ? t('deviceConfig.open') : t('deviceConfig.encryptedNetwork') }}
|
||||
</text>
|
||||
</view>
|
||||
</view>
|
||||
@@ -325,17 +335,17 @@ onMounted(() => {
|
||||
<!-- 密码输入 -->
|
||||
<view v-if="selectedNetwork && selectedNetwork.authmode > 0" class="password-section">
|
||||
<view class="password-item">
|
||||
<text class="password-label">
|
||||
{{ t('deviceConfig.networkPassword') }}
|
||||
</text>
|
||||
<wd-input
|
||||
v-model="password"
|
||||
:placeholder="t('deviceConfig.enterWifiPassword')"
|
||||
show-password
|
||||
clearable
|
||||
@input="onPasswordChange"
|
||||
/>
|
||||
</view>
|
||||
<text class="password-label">
|
||||
{{ t('deviceConfig.wifiPassword') }}
|
||||
</text>
|
||||
<wd-input
|
||||
v-model="password"
|
||||
:placeholder="t('deviceConfig.enterWifiPassword')"
|
||||
show-password
|
||||
clearable
|
||||
@input="onPasswordChange"
|
||||
/>
|
||||
</view>
|
||||
</view>
|
||||
</view>
|
||||
</template>
|
||||
|
||||
@@ -265,7 +265,7 @@ function showAbout() {
|
||||
title: t('settings.aboutApp', { appName: import.meta.env.VITE_APP_TITLE }),
|
||||
content: t('settings.aboutContent', {
|
||||
appName: import.meta.env.VITE_APP_TITLE,
|
||||
version: '0.9.2'
|
||||
version: '0.9.3'
|
||||
}),
|
||||
showCancel: false,
|
||||
confirmText: t('common.confirm'),
|
||||
|
||||
@@ -396,7 +396,28 @@ ASR:
|
||||
type: doubao_stream
|
||||
appid: 你的火山引擎语音合成服务appid
|
||||
access_token: 你的火山引擎语音合成服务access_token
|
||||
cluster: volcengine_input_common
|
||||
# 资源ID:小时版 volc.bigasr.sauc.duration,并发版 volc.bigasr.sauc.concurrent
|
||||
resource_id: volc.bigasr.sauc.duration
|
||||
# 热词、替换词使用流程:https://www.volcengine.com/docs/6561/155738
|
||||
boosting_table_name: (选填)你的热词文件名称
|
||||
correct_table_name: (选填)你的替换词文件名称
|
||||
# 是否开启多语种识别模式
|
||||
enable_multilingual: False
|
||||
# 多语种识别当该键为空时,该模型支持中英文、上海话、闽南语,四川、陕西、粤语识别。当将其设置为特定键时,它可以识别指定语言。
|
||||
# 详细语言列表参考 https://www.volcengine.com/docs/6561/1354869
|
||||
# language: zh-cn
|
||||
# 静音判定时长(ms),默认200ms
|
||||
end_window_size: 200
|
||||
output_dir: tmp/
|
||||
DoubaoStreamASRV2:
|
||||
# 豆包语音识别模型2.0(基于火山引擎seed-asr)
|
||||
# 开通地址:https://console.volcengine.com/speech/service/10038
|
||||
# 价格更为便宜,建议在高并发场景下使用
|
||||
type: doubao_stream
|
||||
appid: 你的火山引擎语音合成服务appid
|
||||
access_token: 你的火山引擎语音合成服务access_token
|
||||
# 资源ID:小时版 volc.seedasr.sauc.duration,并发版 volc.seedasr.sauc.concurrent
|
||||
resource_id: volc.seedasr.sauc.duration
|
||||
# 热词、替换词使用流程:https://www.volcengine.com/docs/6561/155738
|
||||
boosting_table_name: (选填)你的热词文件名称
|
||||
correct_table_name: (选填)你的替换词文件名称
|
||||
@@ -796,6 +817,31 @@ TTS:
|
||||
# - source_speaker: zh_male_ahu_conversation_wvae_bigtts
|
||||
# mix_factor: 0.4
|
||||
# language: "中文" # 指定输出语种,如:中文、英语、日语、韩语等,请根据所选音色支持的语言进行设置,不填则默认为中文
|
||||
#豆包语音合成模型2.0,支持双向流式tts(基于seed-tts-2.0资源)
|
||||
HuoshanDoubleStreamTTSV2:
|
||||
type: huoshan_double_stream
|
||||
# 访问 https://console.volcengine.com/speech/service/10035 开通语音合成大模型,购买音色
|
||||
# 在页面底部获取appid和access_token
|
||||
# 资源ID固定为:seed-tts-2.0(豆包语音合成模型2.0)
|
||||
# 如果是机智云,把接口地址换成wss://bytedance.gizwitsapi.com/api/v3/tts/bidirection
|
||||
# 机智云不需要天填 appid
|
||||
ws_url: wss://openspeech.bytedance.com/api/v3/tts/bidirection
|
||||
appid: 你的火山引擎语音合成服务appid
|
||||
access_token: 你的火山引擎语音合成服务access_token
|
||||
resource_id: seed-tts-2.0
|
||||
speaker: zh_female_xiaohe_uranus_bigtts
|
||||
# 开启WebSocket连接复用,默认复用(注意:复用后设备处于聆听状态时空闲链接会占并发数)
|
||||
enable_ws_reuse: True
|
||||
# 相关参数文档:https://www.volcengine.com/docs/6561/1329505
|
||||
# 音频输出配置(audio_params)- 用户可自定义添加火山引擎支持的任何音频参数
|
||||
audio_params:
|
||||
speech_rate: 0 # 语速(-50~100)
|
||||
loudness_rate: 0 # 音量(-50~100)
|
||||
# 高级文本处理配置(additions)- 用户可自定义添加火山引擎支持的任何高级参数
|
||||
additions:
|
||||
post_process:
|
||||
pitch: 0 # 音高(-12~12)
|
||||
# language: "中文" # 指定输出语种,如:中文、英语、日语、韩语等,请根据所选音色支持的语言进行设置,不填则默认为中文
|
||||
CosyVoiceSiliconflow:
|
||||
type: siliconflow
|
||||
# 硅基流动TTS
|
||||
|
||||
@@ -5,7 +5,7 @@ from config.config_loader import load_config
|
||||
from config.settings import check_config_file
|
||||
from datetime import datetime
|
||||
|
||||
SERVER_VERSION = "0.9.2"
|
||||
SERVER_VERSION = "0.9.3"
|
||||
_logger_initialized = False
|
||||
|
||||
|
||||
|
||||
@@ -116,8 +116,6 @@ class ConnectionHandler:
|
||||
self.memory = _memory
|
||||
self.intent = _intent
|
||||
|
||||
self.is_exiting = False # 标记是否正在执行退出流程
|
||||
|
||||
# 为每个连接单独管理声纹识别
|
||||
self.voiceprint_provider = None
|
||||
|
||||
@@ -312,10 +310,6 @@ class ConnectionHandler:
|
||||
|
||||
async def _route_message(self, message):
|
||||
"""消息路由"""
|
||||
# 退出状态丢弃所有消息
|
||||
if self.is_exiting:
|
||||
return
|
||||
|
||||
# 检查是否已经获取到真实的绑定状态
|
||||
if not self.bind_completed_event.is_set():
|
||||
# 还没有获取到真实状态,等待直到获取到真实状态或超时
|
||||
|
||||
@@ -7,12 +7,9 @@ TAG = __name__
|
||||
|
||||
|
||||
async def handleAbortMessage(conn: "ConnectionHandler"):
|
||||
if conn.close_after_chat or conn.is_exiting:
|
||||
conn.logger.bind(tag=TAG).info("退出流程中被打断,直接关闭连接")
|
||||
return
|
||||
|
||||
conn.logger.bind(tag=TAG).info("Abort message received")
|
||||
# 设置成打断状态,会自动打断llm、tts任务
|
||||
conn.close_after_chat = False
|
||||
conn.client_abort = True
|
||||
conn.clear_queues()
|
||||
# 打断客户端说话状态
|
||||
|
||||
@@ -33,10 +33,6 @@ async def handle_user_intent(conn: "ConnectionHandler", text):
|
||||
if await check_direct_exit(conn, filtered_text):
|
||||
return True
|
||||
|
||||
# 明确再见不被打断
|
||||
if conn.is_exiting:
|
||||
return True
|
||||
|
||||
# 检查是否是唤醒词
|
||||
if await checkWakeupWords(conn, filtered_text):
|
||||
return True
|
||||
@@ -62,7 +58,6 @@ async def check_direct_exit(conn: "ConnectionHandler", text):
|
||||
if text == cmd:
|
||||
conn.logger.bind(tag=TAG).info(f"识别到明确的退出命令: {text}")
|
||||
await send_stt_message(conn, text)
|
||||
conn.is_exiting = True
|
||||
await conn.close()
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -15,8 +15,6 @@ TAG = __name__
|
||||
|
||||
|
||||
async def handleAudioMessage(conn: "ConnectionHandler", audio):
|
||||
if conn.is_exiting:
|
||||
return
|
||||
# 当前片段是否有人说话
|
||||
have_voice = conn.vad.is_vad(conn, audio)
|
||||
# 如果设备刚刚被唤醒,短暂忽略VAD检测
|
||||
|
||||
@@ -84,12 +84,6 @@ class ASRProviderBase(ABC):
|
||||
async def handle_voice_stop(self, conn: "ConnectionHandler", asr_audio_task: List[bytes]):
|
||||
"""并行处理ASR和声纹识别"""
|
||||
try:
|
||||
# 如果处于退出流程中,直接关闭连接,不处理新消息
|
||||
if conn.close_after_chat or conn.is_exiting:
|
||||
logger.bind(tag=TAG).info("退出流程中收到新消息,直接关闭连接")
|
||||
await conn.close()
|
||||
return
|
||||
|
||||
total_start_time = time.monotonic()
|
||||
|
||||
# 准备音频数据
|
||||
|
||||
@@ -26,11 +26,14 @@ class ASRProvider(ASRProviderBase):
|
||||
self.asr_ws = None
|
||||
self.forward_task = None
|
||||
self.is_processing = False # 添加处理状态标志
|
||||
self._is_stopping = False # 添加停止标志,防止竞态条件
|
||||
|
||||
# 配置参数
|
||||
self.appid = str(config.get("appid"))
|
||||
self.cluster = config.get("cluster")
|
||||
self.access_token = config.get("access_token")
|
||||
# 资源ID,用于区分不同的ASR模型(默认1.0模型小时版,v2版本使用seed-asr)
|
||||
self.resource_id = config.get("resource_id", "volc.bigasr.sauc.duration")
|
||||
|
||||
self.boosting_table_name = config.get("boosting_table_name", "")
|
||||
self.correct_table_name = config.get("correct_table_name", "")
|
||||
self.output_dir = config.get("output_dir", "tmp/")
|
||||
@@ -44,7 +47,7 @@ class ASRProvider(ASRProviderBase):
|
||||
if self.enable_multilingual:
|
||||
self.ws_url = "wss://openspeech.bytedance.com/api/v3/sauc/bigmodel_nostream"
|
||||
else:
|
||||
self.ws_url = "wss://openspeech.bytedance.com/api/v3/sauc/bigmodel"
|
||||
self.ws_url = "wss://openspeech.bytedance.com/api/v3/sauc/bigmodel_async"
|
||||
self.uid = config.get("uid", "streaming_asr_service")
|
||||
self.workflow = config.get(
|
||||
"workflow", "audio_in,resample,partition,vad,fe,decode,itn,nlu_punctuate"
|
||||
@@ -146,7 +149,7 @@ class ASRProvider(ASRProviderBase):
|
||||
return
|
||||
|
||||
# 发送当前音频数据
|
||||
if self.asr_ws and self.is_processing:
|
||||
if self.asr_ws and self.is_processing and not self._is_stopping:
|
||||
try:
|
||||
pcm_frame = self.decoder.decode(audio, 960)
|
||||
payload = gzip.compress(pcm_frame)
|
||||
@@ -254,6 +257,7 @@ class ASRProvider(ASRProviderBase):
|
||||
await self.asr_ws.close()
|
||||
self.asr_ws = None
|
||||
self.is_processing = False
|
||||
self._is_stopping = False
|
||||
# 重置所有音频相关状态
|
||||
conn.reset_audio_states()
|
||||
|
||||
@@ -262,9 +266,11 @@ class ASRProvider(ASRProviderBase):
|
||||
asyncio.create_task(self.asr_ws.close())
|
||||
self.asr_ws = None
|
||||
self.is_processing = False
|
||||
self._is_stopping = False
|
||||
|
||||
async def _send_stop_request(self):
|
||||
"""发送最后一个音频帧以通知服务器结束"""
|
||||
self._is_stopping = True # 先标记为停止状态,阻止后续音频发送
|
||||
if self.asr_ws:
|
||||
try:
|
||||
# 发送结束标记的音频帧(gzip压缩的空数据)
|
||||
@@ -283,7 +289,6 @@ class ASRProvider(ASRProviderBase):
|
||||
req = {
|
||||
"app": {
|
||||
"appid": self.appid,
|
||||
"cluster": self.cluster,
|
||||
"token": self.access_token,
|
||||
},
|
||||
"user": {"uid": self.uid},
|
||||
@@ -322,7 +327,7 @@ class ASRProvider(ASRProviderBase):
|
||||
return {
|
||||
"X-Api-App-Key": self.appid,
|
||||
"X-Api-Access-Key": self.access_token,
|
||||
"X-Api-Resource-Id": "volc.bigasr.sauc.duration",
|
||||
"X-Api-Resource-Id": self.resource_id,
|
||||
"X-Api-Connect-Id": str(uuid.uuid4()),
|
||||
}
|
||||
|
||||
@@ -385,9 +390,17 @@ class ASRProvider(ASRProviderBase):
|
||||
"payload_msg": error_msg,
|
||||
}
|
||||
|
||||
# 获取JSON数据(跳过12字节头部)
|
||||
# 获取JSON数据
|
||||
try:
|
||||
json_data = res[12:].decode("utf-8")
|
||||
# 检查字节8-11是否为有效的JSON长度字段
|
||||
# 格式:4字节头 + 4字节序列号 + 4字节长度 + JSON数据
|
||||
length = int.from_bytes(res[8:12], "big")
|
||||
if length > 0 and length <= len(res) - 12:
|
||||
# 有长度字段,从字节12开始读取指定长度的JSON
|
||||
json_data = res[12:12 + length].decode("utf-8")
|
||||
else:
|
||||
# 无长度字段或长度无效,尝试直接解析
|
||||
json_data = res[8:].decode("utf-8")
|
||||
result = json.loads(json_data)
|
||||
logger.bind(tag=TAG).debug(f"成功解析JSON响应: {result}")
|
||||
return {"payload_msg": result}
|
||||
|
||||
@@ -39,13 +39,12 @@ class LLMProvider(LLMProviderBase):
|
||||
}
|
||||
|
||||
# 发起 POST 请求
|
||||
response = requests.post(self.api_url, json=payload, headers=headers)
|
||||
with requests.post(self.api_url, json=payload, headers=headers) as response:
|
||||
# 检查请求是否成功
|
||||
response.raise_for_status()
|
||||
|
||||
# 检查请求是否成功
|
||||
response.raise_for_status()
|
||||
|
||||
# 解析返回数据
|
||||
data = response.json()
|
||||
# 解析返回数据
|
||||
data = response.json()
|
||||
speech = (
|
||||
data.get("response", {})
|
||||
.get("speech", {})
|
||||
|
||||
@@ -50,43 +50,46 @@ class LLMProvider(LLMProviderBase):
|
||||
# 用于处理跨chunk的标签
|
||||
buffer = ""
|
||||
|
||||
for chunk in responses:
|
||||
try:
|
||||
delta = (
|
||||
chunk.choices[0].delta
|
||||
if getattr(chunk, "choices", None)
|
||||
else None
|
||||
)
|
||||
content = delta.content if hasattr(delta, "content") else ""
|
||||
try:
|
||||
for chunk in responses:
|
||||
try:
|
||||
delta = (
|
||||
chunk.choices[0].delta
|
||||
if getattr(chunk, "choices", None)
|
||||
else None
|
||||
)
|
||||
content = delta.content if hasattr(delta, "content") else ""
|
||||
|
||||
if content:
|
||||
# 将内容添加到缓冲区
|
||||
buffer += content
|
||||
if content:
|
||||
# 将内容添加到缓冲区
|
||||
buffer += content
|
||||
|
||||
# 处理缓冲区中的标签
|
||||
while "<think>" in buffer and "</think>" in buffer:
|
||||
# 找到完整的<think></think>标签并移除
|
||||
pre = buffer.split("<think>", 1)[0]
|
||||
post = buffer.split("</think>", 1)[1]
|
||||
buffer = pre + post
|
||||
# 处理缓冲区中的标签
|
||||
while "<think>" in buffer and "</think>" in buffer:
|
||||
# 找到完整的<think></think>标签并移除
|
||||
pre = buffer.split("<think>", 1)[0]
|
||||
post = buffer.split("</think>", 1)[1]
|
||||
buffer = pre + post
|
||||
|
||||
# 处理只有开始标签的情况
|
||||
if "<think>" in buffer:
|
||||
is_active = False
|
||||
buffer = buffer.split("<think>", 1)[0]
|
||||
# 处理只有开始标签的情况
|
||||
if "<think>" in buffer:
|
||||
is_active = False
|
||||
buffer = buffer.split("<think>", 1)[0]
|
||||
|
||||
# 处理只有结束标签的情况
|
||||
if "</think>" in buffer:
|
||||
is_active = True
|
||||
buffer = buffer.split("</think>", 1)[1]
|
||||
# 处理只有结束标签的情况
|
||||
if "</think>" in buffer:
|
||||
is_active = True
|
||||
buffer = buffer.split("</think>", 1)[1]
|
||||
|
||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||
if is_active and buffer:
|
||||
yield buffer
|
||||
buffer = "" # 清空缓冲区
|
||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||
if is_active and buffer:
|
||||
yield buffer
|
||||
buffer = "" # 清空缓冲区
|
||||
|
||||
except Exception as e:
|
||||
logger.bind(tag=TAG).error(f"Error processing chunk: {e}")
|
||||
except Exception as e:
|
||||
logger.bind(tag=TAG).error(f"Error processing chunk: {e}")
|
||||
finally:
|
||||
responses.close()
|
||||
|
||||
def response_with_functions(self, session_id, dialogue, functions=None):
|
||||
# 如果是qwen3模型,在用户最后一条消息中添加/no_think指令
|
||||
@@ -117,49 +120,52 @@ class LLMProvider(LLMProviderBase):
|
||||
is_active = True
|
||||
buffer = ""
|
||||
|
||||
for chunk in stream:
|
||||
try:
|
||||
delta = (
|
||||
chunk.choices[0].delta
|
||||
if getattr(chunk, "choices", None)
|
||||
else None
|
||||
)
|
||||
content = delta.content if hasattr(delta, "content") else None
|
||||
tool_calls = (
|
||||
delta.tool_calls if hasattr(delta, "tool_calls") else None
|
||||
)
|
||||
try:
|
||||
for chunk in stream:
|
||||
try:
|
||||
delta = (
|
||||
chunk.choices[0].delta
|
||||
if getattr(chunk, "choices", None)
|
||||
else None
|
||||
)
|
||||
content = delta.content if hasattr(delta, "content") else None
|
||||
tool_calls = (
|
||||
delta.tool_calls if hasattr(delta, "tool_calls") else None
|
||||
)
|
||||
|
||||
# 如果是工具调用,直接传递
|
||||
if tool_calls:
|
||||
yield None, tool_calls
|
||||
# 如果是工具调用,直接传递
|
||||
if tool_calls:
|
||||
yield None, tool_calls
|
||||
continue
|
||||
|
||||
# 处理文本内容
|
||||
if content:
|
||||
# 将内容添加到缓冲区
|
||||
buffer += content
|
||||
|
||||
# 处理缓冲区中的标签
|
||||
while "<think>" in buffer and "</think>" in buffer:
|
||||
# 找到完整的<think></think>标签并移除
|
||||
pre = buffer.split("<think>", 1)[0]
|
||||
post = buffer.split("</think>", 1)[1]
|
||||
buffer = pre + post
|
||||
|
||||
# 处理只有开始标签的情况
|
||||
if "<think>" in buffer:
|
||||
is_active = False
|
||||
buffer = buffer.split("<think>", 1)[0]
|
||||
|
||||
# 处理只有结束标签的情况
|
||||
if "</think>" in buffer:
|
||||
is_active = True
|
||||
buffer = buffer.split("</think>", 1)[1]
|
||||
|
||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||
if is_active and buffer:
|
||||
yield buffer, None
|
||||
buffer = "" # 清空缓冲区
|
||||
except Exception as e:
|
||||
logger.bind(tag=TAG).error(f"Error processing function chunk: {e}")
|
||||
continue
|
||||
|
||||
# 处理文本内容
|
||||
if content:
|
||||
# 将内容添加到缓冲区
|
||||
buffer += content
|
||||
|
||||
# 处理缓冲区中的标签
|
||||
while "<think>" in buffer and "</think>" in buffer:
|
||||
# 找到完整的<think></think>标签并移除
|
||||
pre = buffer.split("<think>", 1)[0]
|
||||
post = buffer.split("</think>", 1)[1]
|
||||
buffer = pre + post
|
||||
|
||||
# 处理只有开始标签的情况
|
||||
if "<think>" in buffer:
|
||||
is_active = False
|
||||
buffer = buffer.split("<think>", 1)[0]
|
||||
|
||||
# 处理只有结束标签的情况
|
||||
if "</think>" in buffer:
|
||||
is_active = True
|
||||
buffer = buffer.split("</think>", 1)[1]
|
||||
|
||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||
if is_active and buffer:
|
||||
yield buffer, None
|
||||
buffer = "" # 清空缓冲区
|
||||
except Exception as e:
|
||||
logger.bind(tag=TAG).error(f"Error processing function chunk: {e}")
|
||||
continue
|
||||
finally:
|
||||
stream.close()
|
||||
|
||||
@@ -115,21 +115,24 @@ class LLMProvider(LLMProviderBase):
|
||||
responses = self.client.chat.completions.create(**request_params)
|
||||
|
||||
is_active = True
|
||||
for chunk in responses:
|
||||
try:
|
||||
delta = chunk.choices[0].delta if getattr(chunk, "choices", None) else None
|
||||
content = getattr(delta, "content", "") if delta else ""
|
||||
except IndexError:
|
||||
content = ""
|
||||
if content:
|
||||
if "<think>" in content:
|
||||
is_active = False
|
||||
content = content.split("<think>")[0]
|
||||
if "</think>" in content:
|
||||
is_active = True
|
||||
content = content.split("</think>")[-1]
|
||||
if is_active:
|
||||
yield content
|
||||
try:
|
||||
for chunk in responses:
|
||||
try:
|
||||
delta = chunk.choices[0].delta if getattr(chunk, "choices", None) else None
|
||||
content = getattr(delta, "content", "") if delta else ""
|
||||
except IndexError:
|
||||
content = ""
|
||||
if content:
|
||||
if "<think>" in content:
|
||||
is_active = False
|
||||
content = content.split("<think>")[0]
|
||||
if "</think>" in content:
|
||||
is_active = True
|
||||
content = content.split("</think>")[-1]
|
||||
if is_active:
|
||||
yield content
|
||||
finally:
|
||||
responses.close()
|
||||
|
||||
def response_with_functions(self, session_id, dialogue, functions=None, **kwargs):
|
||||
dialogue = self.normalize_dialogue(dialogue)
|
||||
@@ -157,16 +160,19 @@ class LLMProvider(LLMProviderBase):
|
||||
|
||||
stream = self.client.chat.completions.create(**request_params)
|
||||
|
||||
for chunk in stream:
|
||||
if getattr(chunk, "choices", None):
|
||||
delta = chunk.choices[0].delta
|
||||
content = getattr(delta, "content", "")
|
||||
tool_calls = getattr(delta, "tool_calls", None)
|
||||
yield content, tool_calls
|
||||
elif isinstance(getattr(chunk, "usage", None), CompletionUsage):
|
||||
usage_info = getattr(chunk, "usage", None)
|
||||
logger.bind(tag=TAG).info(
|
||||
f"Token 消耗:输入 {getattr(usage_info, 'prompt_tokens', '未知')},"
|
||||
f"输出 {getattr(usage_info, 'completion_tokens', '未知')},"
|
||||
f"共计 {getattr(usage_info, 'total_tokens', '未知')}"
|
||||
)
|
||||
try:
|
||||
for chunk in stream:
|
||||
if getattr(chunk, "choices", None):
|
||||
delta = chunk.choices[0].delta
|
||||
content = getattr(delta, "content", "")
|
||||
tool_calls = getattr(delta, "tool_calls", None)
|
||||
yield content, tool_calls
|
||||
elif isinstance(getattr(chunk, "usage", None), CompletionUsage):
|
||||
usage_info = getattr(chunk, "usage", None)
|
||||
logger.bind(tag=TAG).info(
|
||||
f"Token 消耗:输入 {getattr(usage_info, 'prompt_tokens', '未知')},"
|
||||
f"输出 {getattr(usage_info, 'completion_tokens', '未知')},"
|
||||
f"共计 {getattr(usage_info, 'total_tokens', '未知')}"
|
||||
)
|
||||
finally:
|
||||
stream.close()
|
||||
|
||||
@@ -74,12 +74,15 @@ class LLMProvider(LLMProviderBase):
|
||||
tools=functions,
|
||||
)
|
||||
|
||||
for chunk in stream:
|
||||
delta = chunk.choices[0].delta
|
||||
content = delta.content
|
||||
tool_calls = delta.tool_calls
|
||||
try:
|
||||
for chunk in stream:
|
||||
delta = chunk.choices[0].delta
|
||||
content = delta.content
|
||||
tool_calls = delta.tool_calls
|
||||
|
||||
if content:
|
||||
yield content, tool_calls
|
||||
elif tool_calls:
|
||||
yield None, tool_calls
|
||||
if content:
|
||||
yield content, tool_calls
|
||||
elif tool_calls:
|
||||
yield None, tool_calls
|
||||
finally:
|
||||
stream.close()
|
||||
|
||||
@@ -65,69 +65,8 @@ class BaseASRTester:
|
||||
class DoubaoStreamASRTester(BaseASRTester):
|
||||
def __init__(self):
|
||||
super().__init__("DoubaoStreamASR")
|
||||
|
||||
def _generate_header(
|
||||
self,
|
||||
version=0x01,
|
||||
message_type=0x01,
|
||||
message_type_specific_flags=0x00,
|
||||
serial_method=0x01,
|
||||
compression_type=0x01,
|
||||
reserved_data=0x00,
|
||||
extension_header: bytes = b"",
|
||||
):
|
||||
"""生成协议头(修复:使用正确的Header格式)"""
|
||||
header = bytearray()
|
||||
header_size = int(len(extension_header) / 4) + 1
|
||||
header.append((version << 4) | header_size)
|
||||
header.append((message_type << 4) | message_type_specific_flags)
|
||||
header.append((serial_method << 4) | compression_type)
|
||||
header.append(reserved_data)
|
||||
header.extend(extension_header)
|
||||
return header
|
||||
|
||||
def _generate_audio_default_header(self):
|
||||
"""生成音频数据Header"""
|
||||
return self._generate_header(
|
||||
version=0x01,
|
||||
message_type=0x02,
|
||||
message_type_specific_flags=0x00, # 普通音频帧
|
||||
serial_method=0x01,
|
||||
compression_type=0x01,
|
||||
)
|
||||
|
||||
def _generate_last_audio_header(self):
|
||||
"""生成最后一帧音频的Header(标记音频结束)"""
|
||||
return self._generate_header(
|
||||
version=0x01,
|
||||
message_type=0x02,
|
||||
message_type_specific_flags=0x02, # 0x02表示这是最后一帧
|
||||
serial_method=0x01,
|
||||
compression_type=0x01,
|
||||
)
|
||||
|
||||
def _parse_response(self, res: bytes) -> dict:
|
||||
try:
|
||||
if len(res) < 4:
|
||||
return {"error": "响应数据长度不足"}
|
||||
header = res[:4]
|
||||
message_type = header[1] >> 4
|
||||
if message_type == 0x0F:
|
||||
code = int.from_bytes(res[4:8], "big", signed=False)
|
||||
msg_length = int.from_bytes(res[8:12], "big", signed=False)
|
||||
error_msg = json.loads(res[12:].decode("utf-8"))
|
||||
return {
|
||||
"code": code,
|
||||
"msg_length": msg_length,
|
||||
"payload_msg": error_msg
|
||||
}
|
||||
try:
|
||||
json_data = res[12:].decode("utf-8")
|
||||
return {"payload_msg": json.loads(json_data)}
|
||||
except (UnicodeDecodeError, json.JSONDecodeError):
|
||||
return {"error": "JSON解析失败"}
|
||||
except Exception:
|
||||
return {"error": "解析响应失败"}
|
||||
from core.providers.asr.doubao_stream import ASRProvider as DoubaoStreamProvider
|
||||
self.provider = DoubaoStreamProvider(self.asr_config, delete_audio_file=False)
|
||||
|
||||
async def test(self, test_count=5):
|
||||
if not self.test_audio_files:
|
||||
@@ -138,59 +77,29 @@ class DoubaoStreamASRTester(BaseASRTester):
|
||||
latencies = []
|
||||
for i in range(test_count):
|
||||
try:
|
||||
ws_url = "wss://openspeech.bytedance.com/api/v3/sauc/bigmodel"
|
||||
appid = self.asr_config["appid"]
|
||||
access_token = self.asr_config["access_token"]
|
||||
cluster = self.asr_config.get("cluster", "volcengine_input_common")
|
||||
uid = self.asr_config.get("uid", "streaming_asr_service")
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
headers = {
|
||||
"X-Api-App-Key": appid,
|
||||
"X-Api-Access-Key": access_token,
|
||||
"X-Api-Resource-Id": "volc.bigasr.sauc.duration",
|
||||
"X-Api-Connect-Id": str(uuid.uuid4())
|
||||
}
|
||||
headers = self.provider.token_auth()
|
||||
|
||||
async with websockets.connect(
|
||||
ws_url,
|
||||
self.provider.ws_url,
|
||||
additional_headers=headers,
|
||||
max_size=1000000000,
|
||||
ping_interval=None,
|
||||
ping_timeout=None,
|
||||
close_timeout=10
|
||||
) as ws:
|
||||
request_params = {
|
||||
"app": {"appid": appid, "cluster": cluster, "token": access_token},
|
||||
"user": {"uid": uid},
|
||||
"request": {
|
||||
"reqid": str(uuid.uuid4()),
|
||||
"workflow": "audio_in,resample,partition,vad,fe,decode,itn,nlu_punctuate",
|
||||
"show_utterances": True,
|
||||
"result_type": "single",
|
||||
"sequence": 1
|
||||
},
|
||||
"audio": {
|
||||
"format": "pcm",
|
||||
"codec": "pcm",
|
||||
"rate": 16000,
|
||||
"language": "zh-CN",
|
||||
"bits": 16,
|
||||
"channel": 1,
|
||||
"sample_rate": 16000
|
||||
}
|
||||
}
|
||||
request_params = self.provider.construct_request(str(uuid.uuid4()))
|
||||
|
||||
payload_bytes = str.encode(json.dumps(request_params))
|
||||
payload_bytes = gzip.compress(payload_bytes)
|
||||
full_client_request = self._generate_header()
|
||||
full_client_request = bytearray(self.provider.generate_header())
|
||||
full_client_request.extend((len(payload_bytes)).to_bytes(4, "big"))
|
||||
full_client_request.extend(payload_bytes)
|
||||
await ws.send(full_client_request)
|
||||
|
||||
init_res = await ws.recv()
|
||||
result = self._parse_response(init_res)
|
||||
result = self.provider.parse_response(init_res)
|
||||
if "code" in result and result["code"] != 1000:
|
||||
raise Exception(f"初始化失败: {result.get('payload_msg', {}).get('error', '未知错误')}")
|
||||
|
||||
@@ -200,7 +109,7 @@ class DoubaoStreamASRTester(BaseASRTester):
|
||||
|
||||
# 发送音频数据(使用最后一帧标记,告诉服务端音频已结束)
|
||||
payload = gzip.compress(audio_data)
|
||||
audio_request = bytearray(self._generate_last_audio_header()) # 修复:使用最后一帧Header
|
||||
audio_request = bytearray(self.provider.generate_last_audio_default_header())
|
||||
audio_request.extend(len(payload).to_bytes(4, "big"))
|
||||
audio_request.extend(payload)
|
||||
await ws.send(audio_request)
|
||||
|
||||
@@ -31,12 +31,12 @@ handle_exit_intent_function_desc = {
|
||||
"handle_exit_intent", handle_exit_intent_function_desc, ToolType.SYSTEM_CTL
|
||||
)
|
||||
def handle_exit_intent(conn: "ConnectionHandler", say_goodbye: str | None = None):
|
||||
conn.is_exiting = True
|
||||
# 处理退出意图
|
||||
try:
|
||||
if say_goodbye is None:
|
||||
say_goodbye = "再见,祝您生活愉快!"
|
||||
conn.close_after_chat = True
|
||||
if not conn.close_after_chat:
|
||||
conn.close_after_chat = True
|
||||
logger.bind(tag=TAG).info(f"退出意图已处理:{say_goodbye}")
|
||||
return ActionResponse(
|
||||
action=Action.RESPONSE, result="退出意图已处理", response=say_goodbye
|
||||
|
||||
Reference in New Issue
Block a user