mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-30 18:03:55 +08:00
update:优化百度ASR文档链接
This commit is contained in:
@@ -259,11 +259,8 @@ ASR:
|
|||||||
access_key_secret: 你的阿里云账号access_key_secret
|
access_key_secret: 你的阿里云账号access_key_secret
|
||||||
output_dir: tmp/
|
output_dir: tmp/
|
||||||
BaiduASR:
|
BaiduASR:
|
||||||
# 可以在这里申请百度语音技术的AppID、API Key、Secret Key
|
# 获取AppID、API Key、Secret Key:https://console.bce.baidu.com/ai-engine/old/#/ai/speech/app/list
|
||||||
# https://console.bce.baidu.com/ai-engine/old/#/ai/speech/app/list
|
# 查看资源额度:https://console.bce.baidu.com/ai-engine/old/#/ai/speech/overview/resource/list
|
||||||
# 启用前需要先安装依赖包:
|
|
||||||
# pip install baidu-aip==4.16.13
|
|
||||||
# pip install chardet==5.2.0
|
|
||||||
type: baidu
|
type: baidu
|
||||||
app_id: 你的百度语音技术AppID
|
app_id: 你的百度语音技术AppID
|
||||||
api_key: 你的百度语音技术APIKey
|
api_key: 你的百度语音技术APIKey
|
||||||
|
|||||||
@@ -17,12 +17,16 @@ from config.logger import setup_logging
|
|||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool = True):
|
def __init__(self, config: dict, delete_audio_file: bool = True):
|
||||||
self.app_id = config.get("app_id")
|
self.app_id = config.get("app_id")
|
||||||
self.api_key = config.get("api_key")
|
self.api_key = config.get("api_key")
|
||||||
self.secret_key = config.get("secret_key")
|
self.secret_key = config.get("secret_key")
|
||||||
self.dev_pid = config.get("dev_pid")
|
|
||||||
|
dev_pid = config.get("dev_pid", "1537")
|
||||||
|
self.dev_pid = int(dev_pid) if dev_pid else 1537
|
||||||
|
|
||||||
self.output_dir = config.get("output_dir")
|
self.output_dir = config.get("output_dir")
|
||||||
self.delete_audio_file = delete_audio_file
|
self.delete_audio_file = delete_audio_file
|
||||||
|
|
||||||
@@ -61,7 +65,9 @@ class ASRProvider(ASRProviderBase):
|
|||||||
|
|
||||||
return pcm_data
|
return pcm_data
|
||||||
|
|
||||||
async def speech_to_text(self, opus_data: List[bytes], session_id: str) -> Tuple[Optional[str], Optional[str]]:
|
async def speech_to_text(
|
||||||
|
self, opus_data: List[bytes], session_id: str
|
||||||
|
) -> Tuple[Optional[str], Optional[str]]:
|
||||||
"""将语音数据转换为文本"""
|
"""将语音数据转换为文本"""
|
||||||
if not opus_data:
|
if not opus_data:
|
||||||
logger.bind(tag=TAG).warn("音频数据为空!")
|
logger.bind(tag=TAG).warn("音频数据为空!")
|
||||||
@@ -86,16 +92,25 @@ class ASRProvider(ASRProviderBase):
|
|||||||
|
|
||||||
start_time = time.time()
|
start_time = time.time()
|
||||||
# 识别本地文件
|
# 识别本地文件
|
||||||
result = self.client.asr(combined_pcm_data, 'pcm', 16000, {
|
result = self.client.asr(
|
||||||
'dev_pid': str(self.dev_pid),
|
combined_pcm_data,
|
||||||
})
|
"pcm",
|
||||||
|
16000,
|
||||||
|
{
|
||||||
|
"dev_pid": str(self.dev_pid),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
if result and result["err_no"] == 0:
|
if result and result["err_no"] == 0:
|
||||||
logger.bind(tag=TAG).debug(f"百度语音识别耗时: {time.time() - start_time:.3f}s | 结果: {result}")
|
logger.bind(tag=TAG).debug(
|
||||||
|
f"百度语音识别耗时: {time.time() - start_time:.3f}s | 结果: {result}"
|
||||||
|
)
|
||||||
result = result["result"][0]
|
result = result["result"][0]
|
||||||
return result, file_path
|
return result, file_path
|
||||||
else:
|
else:
|
||||||
raise Exception(f"百度语音识别失败,错误码: {result['err_no']},错误信息: {result['err_msg']}")
|
raise Exception(
|
||||||
|
f"百度语音识别失败,错误码: {result['err_no']},错误信息: {result['err_msg']}"
|
||||||
|
)
|
||||||
return None, file_path
|
return None, file_path
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
Reference in New Issue
Block a user