diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..19b6976 --- /dev/null +++ b/.gitignore @@ -0,0 +1,18 @@ +# 音频文件 +*.pcm +*.mp3 + +# Python +__pycache__/ +*.py[cod] +*$py.class + +# 环境文件 +.env +.venv +env/ +venv/ + +# IDE +.idea/ +.vscode/ \ No newline at end of file diff --git a/README.md b/README.md index bb14c2b..5ef4f11 100644 --- a/README.md +++ b/README.md @@ -8,19 +8,127 @@ 2. 开通以下服务: - 短语音识别服务 - 短文本在线合成服务 - + - 其它的语音库根据需求自行开通 +## 安装方式 + 将baidu_voice目录复制到HomeAssistant的config/custom_components目录下,重新启动HomeAssistant即可。 ## 配置步骤 -1. 在百度云控制台创建应用并获取 API Key 和 Secret Key +1. 在百度云控制台创建应用并获取 APP ID、API Key 和 Secret Key 2. 在 Home Assistant 中添加 baidu_voice 集成 -3. 输入您的 API Key 和 Secret Key - +3. 输入您的 APP ID、API Key 和 Secret Key,输入其它的默认设置参数 ## 配置截图 -![百度tts语音服务配置](baidu1.png) -![百度stt语音服务配置](baidu2.png) +![百度tts语音服务配置](settings.png) + + +##参考资料 + +[短语音识别标准版API](https://ai.baidu.com/ai-doc/SPEECH/Jlbxdezuf) + +[短文本在线合成API](https://ai.baidu.com/ai-doc/SPEECH/mlbxh7xie) ## 注意事项 - 本集成需要互联网连接 - 使用百度语音服务可能会产生费用,请参考百度云官方计费标准 + + + + +## 百度智能云服务开通界面 + +![百度tts语音服务配置](baidu1.png) +![百度stt语音服务配置](baidu2.png) + +## 百度语音合成音色列表 +[TTS全部音色查询](https://ai.baidu.com/ai-doc/SPEECH/Rluv3uq3d) + +### 基础音库 + +| 音色ID | 音色名称 | 性别 | 描述 | +|--------|----------|------|------| +| 0 | 度小美 | 女 | 标准女主播 | +| 1 | 度小宇 | 男 | 亲切男声 | +| 3 | 度逍遥 | 男 | 情感男声 | +| 4 | 度丫丫 | 女 | 童声 | +| 5 | 度小娇 | 女 | 成熟女主播 | + +### 精品音库 + +| 音色ID | 音色名称 | 性别 | 描述 | +|--------|----------|------|------| +| 5003 | 度逍遥 | 男 | 情感男声 | +| 5118 | 度小鹿 | 女 | 甜美女声 | +| 106 | 度博文 | 男 | 专业男主播 | +| 103 | 度米朵 | 女 | 可爱童声 | +| 110 | 度小童 | 女 | 童声主播 | +| 111 | 度小萌 | 女 | 软萌妹子 | + +### 臻品音库 + +| 音色ID | 音色名称 | 性别 | 描述 | +|--------|----------|------|------| +| 4003 | 度逍遥 | 男 | 情感男声 | +| 4106 | 度博文 | 男 | 专业男主播 | +| 4115 | 度小贤 | 男 | 电台男主播 | +| 5147 | 度常盈 | 女 | 电台女主播 | +| 5976 | 度小皮 | 女 | 萌娃童声 | +| 5971 | 度皮特 | 男 | 老外男声 | +| 4164 | 度阿肯 | 男 | 主播男声 | +| 4176 | 度有为 | 男 | 磁性男声 | +| 4259 | 度小新 | 女 | 播音女声 | +| 4119 | 度小鹿 | 女 | 甜美女声 | +| 4105 | 度灵儿 | 女 | 清激女声 | +| 4117 | 度小乔 | 女 | 活泼女声 | +| 4288 | 度晴岚 | 女 | 甜美女声 | +| 4192 | 度青川 | 男 | 温柔男声 | +| 4100 | 度小雯 | 女 | 活力女主播 | +| 4103 | 度米朵 | 女 | 可爱女声 | +| 4144 | 度姗姗 | 女 | 娱乐女声 | +| 4278 | 度小贝 | 女 | 知识女主播 | +| 4143 | 度清风 | 男 | 配音男声 | +| 4140 | 度小新 | 女 | 专业女主播 | +| 4129 | 度小彦 | 男 | 知识男主播 | +| 4149 | 度星河 | 男 | 广告男声 | +| 4254 | 度小清 | 女 | 广告女声 | +| 4206 | 度博文 | 男 | 综艺男声 | +| 4147 | 度云朵 | 女 | 可爱童声 | +| 4141 | 度婉婉 | 女 | 甜美女声 | +| 4226 | 南方 | 女 | 电台女主播 | +| 6205 | 度悠然 | 男 | 旁白男声 | +| 6221 | 度云萱 | 女 | 旁白女声 | +| 6546 | 度清豪 | 男 | 逍遥侠客 | +| 6602 | 度清柔 | 男 | 温柔男神 | +| 6562 | 度雨楠 | 女 | 元气少女 | +| 6543 | 度雨萌 | 女 | 邻家女孩 | +| 6747 | 度书古 | 男 | 情感男声 | +| 6748 | 度书严 | 男 | 沉稳男声 | +| 6746 | 度书道 | 男 | 沉稳男声 | +| 6644 | 度书宁 | 女 | 亲和女声 | +| 4148 | 度小夏 | 女 | 甜美女声 | +| 4277 | 西贝 | 女 | 脱口秀女声 | +| 4114 | 阿龙 | 男 | 说书男声 | + +### 大模型音库 + +| 音色ID | 音色名称 | 性别 | 描述 | +|--------|----------|------|------| +| 4179 | 度泽言 | 男 | 温暖男声 | +| 4146 | 度禧禧 | 女 | 阳光女声 | +| 6567 | 度小柔 | 女 | 温柔女声 | +| 4189 | 度涵竹 | 女 | 温暖女声 | +| 4194 | 度嫣然 | 女 | 甜美主播 | +| 4193 | 度泽言 | 男 | 温暖男声 | +| 4195 | 度怀安 | 男 | 沉稳男声 | +| 4196 | 度清影 | 女 | 清亮女声 | +| 4197 | 度沁遥 | 女 | 温柔女声 | +| 20100 | 度小粤 | 女 | 粤语女声 | +| 20101 | 度晓芸 | 女 | 粤语女声 | +| 4257 | 四川小哥 | 男 | 四川方言 | +| 4132 | 度阿闽 | 男 | 闽南语男声 | +| 4139 | 度小蓉 | 女 | 四川方言 | +| 5977 | 台媒女声 | 女 | 台湾腔 | +| 4007 | 度小台 | 女 | 台湾腔 | +| 4150 | 度湘玉 | 女 | 湖南方言 | +| 4134 | 度阿锦 | 男 | 粤语男声 | +| 4172 | 度筱林 | 女 | 温柔女声 | diff --git a/config_flow.py b/config_flow.py deleted file mode 100644 index 0f1a3db..0000000 --- a/config_flow.py +++ /dev/null @@ -1,178 +0,0 @@ -"""百度语音配置流程.""" -import logging -import os -import asyncio -import voluptuous as vol -from typing import Any, Dict, Optional - -from homeassistant import config_entries -from homeassistant.core import callback -from homeassistant.data_entry_flow import FlowResult -from homeassistant.helpers import config_validation as cv -from aip import AipSpeech - -from .const import ( - DOMAIN, - CONF_APP_ID, - CONF_API_KEY, - CONF_SECRET_KEY, -) - -_LOGGER = logging.getLogger(__name__) - -# 测试音频文件路径 -TEST_AUDIO_PATH = os.path.join(os.path.dirname(__file__), "text2audio.pcm") - -async def async_read_file(file_path: str) -> bytes: - """异步读取文件内容.""" - loop = asyncio.get_event_loop() - return await loop.run_in_executor(None, lambda: open(file_path, "rb").read()) - -async def async_validate_api( - app_id: str, - api_key: str, - secret_key: str, -) -> bool: - """验证百度API配置.""" - try: - # 创建语音识别客户端 - client = AipSpeech(app_id, api_key, secret_key) - - # 测试TTS - result = await asyncio.get_event_loop().run_in_executor( - None, - client.synthesis, - "测试语音合成", - "zh", - 1, # 1表示mp3格式 - { - "vol": 5, # 音量,取值0-15,默认为5中音量 - "per": 0, # 发音人选择,0为女声,1为男声,3为情感合成-度逍遥,4为情感合成-度丫丫 - "spd": 5, # 语速,取值0-9,默认为5中语速 - "pit": 5, # 音调,取值0-9,默认为5中语调 - } - ) - - if isinstance(result, dict): - _LOGGER.error("TTS测试失败: %s", result.get('err_msg')) - return False - - # 测试STT - try: - audio_data = await async_read_file(TEST_AUDIO_PATH) - except Exception as ex: - _LOGGER.error("无法读取测试音频文件: %s", ex) - return False - - result = await asyncio.get_event_loop().run_in_executor( - None, - client.asr, - audio_data, - "pcm", # PCM格式 - 16000, # 采样率 - { - "dev_pid": 1537, # 普通话(支持简单的英文识别) - } - ) - - if result.get("err_no") != 0: - _LOGGER.error("STT测试失败: %s", result.get('err_msg')) - return False - - return True - - except Exception as ex: - _LOGGER.error("API验证异常: %s", ex) - return False - -class BaiduVoiceConfigFlow(config_entries.ConfigFlow, domain=DOMAIN): - """处理百度语音配置流程.""" - - VERSION = 1 - CONNECTION_CLASS = config_entries.CONN_CLASS_CLOUD_PUSH - - async def async_step_user( - self, user_input: Optional[Dict[str, Any]] = None - ) -> FlowResult: - """处理用户初始步骤.""" - errors: Dict[str, str] = {} - - if user_input is not None: - # 验证API配置 - if await async_validate_api( - user_input[CONF_APP_ID], - user_input[CONF_API_KEY], - user_input[CONF_SECRET_KEY], - ): - return self.async_create_entry( - title="百度语音", - data=user_input, - ) - errors["base"] = "invalid_auth" - - return self.async_show_form( - step_id="user", - data_schema=vol.Schema( - { - vol.Required(CONF_APP_ID): str, - vol.Required(CONF_API_KEY): str, - vol.Required(CONF_SECRET_KEY): str, - } - ), - errors=errors, - ) - - @staticmethod - @callback - def async_get_options_flow( - config_entry: config_entries.ConfigEntry, - ) -> config_entries.OptionsFlow: - """创建选项流.""" - return BaiduVoiceOptionsFlow(config_entry) - -class BaiduVoiceOptionsFlow(config_entries.OptionsFlow): - """处理百度语音选项.""" - - def __init__(self, config_entry: config_entries.ConfigEntry) -> None: - """初始化选项流.""" - self._config_entry = config_entry - - async def async_step_init( - self, user_input: Optional[Dict[str, Any]] = None - ) -> FlowResult: - """管理百度语音选项.""" - errors: Dict[str, str] = {} - - if user_input is not None: - # 验证API配置 - if await async_validate_api( - user_input[CONF_APP_ID], - user_input[CONF_API_KEY], - user_input[CONF_SECRET_KEY], - ): - return self.async_create_entry( - title="", - data=user_input, - ) - errors["base"] = "invalid_auth" - - return self.async_show_form( - step_id="init", - data_schema=vol.Schema( - { - vol.Required( - CONF_APP_ID, - default=self._config_entry.data.get(CONF_APP_ID), - ): str, - vol.Required( - CONF_API_KEY, - default=self._config_entry.data.get(CONF_API_KEY), - ): str, - vol.Required( - CONF_SECRET_KEY, - default=self._config_entry.data.get(CONF_SECRET_KEY), - ): str, - } - ), - errors=errors, - ) \ No newline at end of file diff --git a/const.py b/const.py deleted file mode 100644 index ab9b558..0000000 --- a/const.py +++ /dev/null @@ -1,45 +0,0 @@ -"""百度语音常量.""" -from typing import Final - -DOMAIN: Final = "baidu_voice" - -# 配置项 -CONF_APP_ID: Final = "app_id" -CONF_API_KEY: Final = "api_key" -CONF_SECRET_KEY: Final = "secret_key" - -# 服务 -SERVICE_TTS: Final = "tts" -SERVICE_STT: Final = "stt" - -# 服务属性 -ATTR_MESSAGE: Final = "message" -ATTR_LANGUAGE: Final = "language" -ATTR_VOLUME: Final = "volume" -ATTR_SPEED: Final = "speed" -ATTR_PITCH: Final = "pitch" -ATTR_PERSON: Final = "person" -ATTR_AUDIO: Final = "audio" -ATTR_FORMAT: Final = "format" -ATTR_SAMPLE_RATE: Final = "sample_rate" - -# 语音参数 -DEFAULT_SPEED: Final = 5 -DEFAULT_PITCH: Final = 5 -DEFAULT_VOLUME: Final = 5 -DEFAULT_PERSON: Final = 0 # 0: 女声, 1: 男声, 3: 情感男声, 4: 情感女声 - -# 音频格式 -SUPPORTED_AUDIO_FORMATS: Final = ["mp3", "wav", "pcm"] -SUPPORTED_SAMPLE_RATES: Final = [8000, 16000] -SUPPORTED_CHANNELS: Final = [1] # 单声道 -SUPPORTED_CODECS: Final = ["pcm", "mp3", "wav"] - -# 语言支持 -SUPPORTED_LANGUAGES: Final = ["zh"] # 目前只支持中文 - -# 错误码 -ERROR_INVALID_AUTH: Final = "invalid_auth" -ERROR_UNKNOWN: Final = "unknown" -ERROR_NOT_SUPPORTED: Final = "not_supported" -ERROR_SERVICE_UNAVAILABLE: Final = "service_unavailable" \ No newline at end of file diff --git a/__init__.py b/custom_components/baidu_voice/__init__.py similarity index 56% rename from __init__.py rename to custom_components/baidu_voice/__init__.py index 09f3dc6..2893852 100644 --- a/__init__.py +++ b/custom_components/baidu_voice/__init__.py @@ -1,49 +1,40 @@ -"""百度语音集成.""" +"""Integration for Baidu Voice services.""" + import logging -from typing import Any, Dict from homeassistant.config_entries import ConfigEntry from homeassistant.core import HomeAssistant -from homeassistant.helpers import config_validation as cv -from aip import AipSpeech +from homeassistant.helpers.typing import ConfigType -from .const import ( - DOMAIN, - CONF_APP_ID, - CONF_API_KEY, - CONF_SECRET_KEY, -) +from .const import DOMAIN _LOGGER = logging.getLogger(__name__) -CONFIG_SCHEMA = cv.deprecated(DOMAIN) +PLATFORMS = ["stt", "tts"] -PLATFORMS = ["tts", "stt"] -async def async_setup(hass: HomeAssistant, config: Dict[str, Any]) -> bool: +async def async_setup(hass: HomeAssistant, config: ConfigType) -> bool: """设置百度语音集成.""" return True + async def async_setup_entry(hass: HomeAssistant, entry: ConfigEntry) -> bool: """设置百度语音配置项.""" # 存储配置数据 hass.data.setdefault(DOMAIN, {}) - hass.data[DOMAIN] = { - CONF_APP_ID: entry.data[CONF_APP_ID], - CONF_API_KEY: entry.data[CONF_API_KEY], - CONF_SECRET_KEY: entry.data[CONF_SECRET_KEY], - } - + hass.data[DOMAIN] = entry.data.copy() # 保存所有配置数据 + # 加载平台 await hass.config_entries.async_forward_entry_setups(entry, PLATFORMS) return True + async def async_unload_entry(hass: HomeAssistant, entry: ConfigEntry) -> bool: """卸载百度语音配置项.""" # 卸载平台 unload_ok = await hass.config_entries.async_unload_platforms(entry, PLATFORMS) - + if unload_ok: hass.data.pop(DOMAIN) - - return unload_ok \ No newline at end of file + + return unload_ok diff --git a/custom_components/baidu_voice/config_flow.py b/custom_components/baidu_voice/config_flow.py new file mode 100644 index 0000000..7de9778 --- /dev/null +++ b/custom_components/baidu_voice/config_flow.py @@ -0,0 +1,170 @@ +"""Config flow for Baidu TTS integration.""" + +from __future__ import annotations + +import logging +from typing import Any + +import voluptuous as vol + +from homeassistant.config_entries import ( + ConfigEntry, + ConfigFlow, + ConfigFlowResult, + OptionsFlow, +) +from homeassistant.core import callback + +from .const import ( + CONF_API_KEY, + CONF_APP_ID, + CONF_SECRET_KEY, + DOMAIN, + STT_CONF_LANGUAGE, + STT_DEFAULT_LANGUAGE, + STT_LANGUAGES, + TTS_CONF_FILEFORMAT, + TTS_CONF_LANGUAGE, + TTS_CONF_PITCH, + TTS_CONF_SPEED, + TTS_CONF_VOICE, + TTS_CONF_VOLUME, + TTS_DEFAULT_FILEFORMAT, + TTS_DEFAULT_LANGUAGE, + TTS_DEFAULT_PITCH, + TTS_DEFAULT_SPEED, + TTS_DEFAULT_VOICE, + TTS_DEFAULT_VOLUME, + TTS_FILEFORMAT_MAP, + TTS_LANGUAGES, + TTS_SUPPORTED_VOICES, +) + +_LOGGER = logging.getLogger(__name__) + +STEP_USER_DATA_SCHEMA = vol.Schema( + { + vol.Required(CONF_APP_ID): str, + vol.Required(CONF_API_KEY): str, + vol.Required(CONF_SECRET_KEY): str, + vol.Required(TTS_CONF_LANGUAGE, default=TTS_DEFAULT_LANGUAGE): vol.In( + TTS_LANGUAGES + ), + vol.Optional(TTS_CONF_SPEED, default=TTS_DEFAULT_SPEED): vol.All( + vol.Coerce(int), vol.Range(min=0, max=9) + ), + vol.Optional(TTS_CONF_PITCH, default=TTS_DEFAULT_PITCH): vol.All( + vol.Coerce(int), vol.Range(min=0, max=9) + ), + vol.Optional(TTS_CONF_VOLUME, default=TTS_DEFAULT_VOLUME): vol.All( + vol.Coerce(int), vol.Range(min=0, max=15) + ), + vol.Optional(TTS_CONF_VOICE, default=TTS_DEFAULT_VOICE): vol.In( + TTS_SUPPORTED_VOICES + ), + vol.Optional(STT_CONF_LANGUAGE, default=STT_DEFAULT_LANGUAGE): vol.In( + STT_LANGUAGES + ), + } +) + + +class BaiduTTSConfigFlow(ConfigFlow, domain=DOMAIN): + """Handle a config flow for Baidu TTS.""" + + VERSION = 1 + + async def async_step_user( + self, user_input: dict[str, Any] | None = None + ) -> ConfigFlowResult: + """Handle the initial step.""" + if user_input is not None: + # Provide unique ID to prevent duplicate entries + app_id = user_input[CONF_APP_ID] + await self.async_set_unique_id(f"baidu_tts_{app_id}") + self._abort_if_unique_id_configured() + + return self.async_create_entry(title="百度语音合成", data=user_input) + + return self.async_show_form(step_id="user", data_schema=STEP_USER_DATA_SCHEMA) + + @classmethod + @callback + def async_get_options_flow(cls, config_entry: ConfigEntry) -> BaiduTTSOptionsFlow: + """Get the options flow for this handler.""" + return BaiduTTSOptionsFlow() + + +class BaiduTTSOptionsFlow(OptionsFlow): + """Baidu TTS integration options handler.""" + + async def async_step_init( + self, user_input: dict[str, Any] | None = None + ) -> ConfigFlowResult: + """Manage the options.""" + if user_input is not None: + return self.async_create_entry(title="", data=user_input) + + options = self.config_entry.options + data = self.config_entry.data + + schema = { + vol.Required( + CONF_APP_ID, + default=options.get(CONF_APP_ID, data.get(CONF_APP_ID)), + ): str, + vol.Required( + CONF_API_KEY, + default=options.get(CONF_API_KEY, data.get(CONF_API_KEY)), + ): str, + vol.Required( + CONF_SECRET_KEY, + default=options.get(CONF_SECRET_KEY, data.get(CONF_SECRET_KEY)), + ): str, + vol.Required( + TTS_CONF_LANGUAGE, + default=options.get(TTS_CONF_LANGUAGE, data.get(TTS_CONF_LANGUAGE)), + ): vol.In(TTS_LANGUAGES), + vol.Optional( + TTS_CONF_SPEED, + default=options.get( + TTS_CONF_SPEED, data.get(TTS_CONF_SPEED, TTS_DEFAULT_SPEED) + ), + ): vol.All(vol.Coerce(int), vol.Range(min=0, max=9)), + vol.Optional( + TTS_CONF_PITCH, + default=options.get( + TTS_CONF_PITCH, data.get(TTS_CONF_PITCH, TTS_DEFAULT_PITCH) + ), + ): vol.All(vol.Coerce(int), vol.Range(min=0, max=9)), + vol.Optional( + TTS_CONF_VOLUME, + default=options.get( + TTS_CONF_VOLUME, data.get(TTS_CONF_VOLUME, TTS_DEFAULT_VOLUME) + ), + ): vol.All(vol.Coerce(int), vol.Range(min=0, max=15)), + vol.Optional( + TTS_CONF_VOICE, + default=options.get( + TTS_CONF_VOICE, data.get(TTS_CONF_VOICE, TTS_DEFAULT_VOICE) + ), + ): vol.In(TTS_SUPPORTED_VOICES), + vol.Optional( + TTS_CONF_FILEFORMAT, + default=options.get( + TTS_CONF_FILEFORMAT, + data.get(TTS_CONF_FILEFORMAT, TTS_DEFAULT_FILEFORMAT), + ), + ): vol.In(TTS_FILEFORMAT_MAP), + vol.Optional( + STT_CONF_LANGUAGE, + default=options.get( + STT_CONF_LANGUAGE, data.get(STT_CONF_LANGUAGE, STT_DEFAULT_LANGUAGE) + ), + ): vol.In(STT_LANGUAGES), + } + + return self.async_show_form( + step_id="init", + data_schema=vol.Schema(schema), + ) diff --git a/custom_components/baidu_voice/const.py b/custom_components/baidu_voice/const.py new file mode 100644 index 0000000..baec114 --- /dev/null +++ b/custom_components/baidu_voice/const.py @@ -0,0 +1,141 @@ +"""百度语音常量定义.""" + +from typing import Final + +DOMAIN: Final = "baidu_voice" + +# 配置项 +CONF_APP_ID: Final = "app_id" +CONF_API_KEY: Final = "api_key" +CONF_SECRET_KEY: Final = "secret_key" + +# 通用选项 +TTS_CONF_LANGUAGE: Final = "language" +TTS_CONF_VOICE: Final = "voice" +TTS_CONF_VOLUME: Final = "volume" +TTS_CONF_SPEED: Final = "speed" +TTS_CONF_PITCH: Final = "pitch" +TTS_CONF_FILEFORMAT: Final = "fileformat" + +# TTS语言选项 +TTS_LANGUAGES: Final = {"zh": "简体中文", "en": "English"} +TTS_DEFAULT_LANGUAGE: Final = "zh" + +# TTS格式映射 +TTS_FILEFORMAT_MAP: Final = { + 3: "mp3", # mp3-16k/24k + 4: "pcm", # pcm-16k/24k + 5: "pcm8k", # pcm-8k + 6: "wav", # wav(同pcm-16k/24k) +} + +# 定义支持的语音选项 +TTS_SUPPORTED_VOICES = { + 0: "度小美-标准女主播", + 1: "度小宇-亲切男声", + 3: "度逍遥-情感男声", + 4: "度丫丫-童声", + 5: "度小娇-成熟女主播", + 5003: "度逍遥-情感男声", + 5118: "度小鹿-甜美女声", + 106: "度博文-专业男主播", + 103: "度米朵-可爱童声", + 110: "度小童-童声主播", + 111: "度小萌-软萌妹子", + 4003: "度逍遥-情感男声", + 4106: "度博文-专业男主播", + 4115: "度小贤-电台男主播", + 5147: "度常盈-电台女主播", + 5976: "度小皮-萌娃童声", + 5971: "度皮特-老外男声", + 4164: "度阿肯-主播男声", + 4176: "度有为-磁性男声", + 4259: "度小新-播音女声", + 4119: "度小鹿-甜美女声", + 4105: "度灵儿-清激女声", + 4117: "度小乔-活泼女声", + 4288: "度晴岚-甜美女声", + 4192: "度青川-温柔男声", + 4100: "度小雯-活力女主播", + 4103: "度米朵-可爱女声", + 4144: "度姗姗-娱乐女声", + 4278: "度小贝-知识女主播", + 4143: "度清风-配音男声", + 4140: "度小新-专业女主播", + 4129: "度小彦-知识男主播", + 4149: "度星河-广告男声", + 4254: "度小清-广告女声", + 4206: "度博文-综艺男声", + 4147: "度云朵-可爱童声", + 4141: "度婉婉-甜美女声", + 4226: "南方-电台女主播", + 6205: "度悠然-旁白男声", + 6221: "度云萱-旁白女声", + 6546: "度清豪-逍遥侠客", + 6602: "度清柔-温柔男神", + 6562: "度雨楠-元气少女", + 6543: "度雨萌-邻家女孩", + 6747: "度书古-情感男声", + 6748: "度书严-沉稳男声", + 6746: "度书道-沉稳男声", + 6644: "度书宁-亲和女声", + 4148: "度小夏-甜美女声", + 4277: "西贝-脱口秀女声", + 4114: "阿龙-说书男声", + 4179: "度泽言-温暖男声", + 4146: "度禧禧-阳光女声", + 6567: "度小柔-温柔女声", + 4156: "度言浩-年轻男声", + 4189: "度涵竹-开朗女声", + 4194: "度嫣然-活泼女声", + 4193: "度泽言-开朗男声", + 4195: "度怀安-磁性男声", + 4196: "度清影-甜美女声", + 4197: "度沁遥-知性女声", + 20100: "度小粤-粤语女声", + 20101: "度晓芸-粤语女声", + 4257: "四川小哥-四川男声", + 4132: "度阿闽-闽南男声", + 4139: "度小蓉-四川女声", + 5977: "台媒女声-台湾女声", + 4007: "度小台-台湾女声", + 4150: "度湘玉-陕西女声", + 4134: "度阿锦-东北女声", + 4172: "度筱林-天津女声", + 5980: "度阿花-上海女声", + 4154: "度老崔-北京男声", +} + +# 默认值 +TTS_DEFAULT_VOLUME: Final = 5 +TTS_DEFAULT_SPEED: Final = 5 +TTS_DEFAULT_PITCH: Final = 5 +TTS_DEFAULT_VOICE: Final = 0 +TTS_DEFAULT_FILEFORMAT: Final = 3 # 默认音频格式 MP3 + +# 范围限制 +TTS_VOLUME_RANGE: Final = (0, 15) +TTS_SPEED_RANGE: Final = (0, 9) +TTS_PITCH_RANGE: Final = (0, 9) + +# 语言选项 +STT_LANGUAGES: Final = { + "zh-CN": "普通话", + "en-US": "英语", + "zh-HK": "粤语", + "zh-TW": "四川话", +} +STT_LANGUAGES_CODE_MAP: Final = { + "zh-CN": 1537, + "en-US": 1737, + "zh-HK": 1637, + "zh-TW": 1837, +} +# STT配置项 +STT_CONF_LANGUAGE: Final = "stt_language" + +# STT语言选项 +STT_DEFAULT_LANGUAGE: Final = "zh-CN" # 默认使用普通话 + +# 错误码 +ERROR_INVALID_AUTH: Final = "invalid_auth" diff --git a/custom_components/baidu_voice/hacs.json b/custom_components/baidu_voice/hacs.json new file mode 100644 index 0000000..c25719e --- /dev/null +++ b/custom_components/baidu_voice/hacs.json @@ -0,0 +1,6 @@ +{ + "name": "Baidu Voice", + "domains": ["tts", "stt"], + "homeassistant": "2023.1.0", + "render_readme": true +} \ No newline at end of file diff --git a/icon.png b/custom_components/baidu_voice/icon.png similarity index 100% rename from icon.png rename to custom_components/baidu_voice/icon.png diff --git a/custom_components/baidu_voice/info.md b/custom_components/baidu_voice/info.md new file mode 100644 index 0000000..b481f62 --- /dev/null +++ b/custom_components/baidu_voice/info.md @@ -0,0 +1,19 @@ +# 百度语音服务集成 + +这是一个 Home Assistant 集成,用于接入百度语音服务,提供文本转语音(TTS)和语音转文本(STT)功能。 + +## 主要功能 + +- 文本转语音(TTS) +- 语音转文本(STT) +- 支持多种音色选择 +- 支持多种音频格式 +- 支持多种语言 + +## 配置要求 + +1. 需要在百度云创建应用 +2. 开通以下服务: + - 短语音识别服务 + - 短文本在线合成服务 + - 其它的语音库根据需求自行开通 \ No newline at end of file diff --git a/custom_components/baidu_voice/manifest.json b/custom_components/baidu_voice/manifest.json new file mode 100644 index 0000000..2bdfd4e --- /dev/null +++ b/custom_components/baidu_voice/manifest.json @@ -0,0 +1,17 @@ +{ + "domain": "baidu_voice", + "name": "Baidu Voice", + "codeowners": [ + "@howelljiang" + ], + "config_flow": true, + "dependencies": [], + "documentation": "https://github.com/howelljiang/baidu-voice", + "iot_class": "cloud_push", + "quality_scale": "bronze", + "requirements": [ + "baidu-aip == 4.16.10", "chardet==5.2.0" + ], + "version": "0.0.1", + "issue_tracker": "https://github.com/howelljiang/baidu-voice/issues" +} diff --git a/custom_components/baidu_voice/quality_scale.yaml b/custom_components/baidu_voice/quality_scale.yaml new file mode 100644 index 0000000..201a916 --- /dev/null +++ b/custom_components/baidu_voice/quality_scale.yaml @@ -0,0 +1,60 @@ +rules: + # Bronze + action-setup: todo + appropriate-polling: todo + brands: todo + common-modules: todo + config-flow-test-coverage: todo + config-flow: todo + dependency-transparency: todo + docs-actions: todo + docs-high-level-description: todo + docs-installation-instructions: todo + docs-removal-instructions: todo + entity-event-setup: todo + entity-unique-id: todo + has-entity-name: todo + runtime-data: todo + test-before-configure: todo + test-before-setup: todo + unique-config-entry: todo + + # Silver + action-exceptions: todo + config-entry-unloading: todo + docs-configuration-parameters: todo + docs-installation-parameters: todo + entity-unavailable: todo + integration-owner: todo + log-when-unavailable: todo + parallel-updates: todo + reauthentication-flow: todo + test-coverage: todo + + # Gold + devices: todo + diagnostics: todo + discovery-update-info: todo + discovery: todo + docs-data-update: todo + docs-examples: todo + docs-known-limitations: todo + docs-supported-devices: todo + docs-supported-functions: todo + docs-troubleshooting: todo + docs-use-cases: todo + dynamic-devices: todo + entity-category: todo + entity-device-class: todo + entity-disabled-by-default: todo + entity-translations: todo + exception-translations: todo + icon-translations: todo + reconfiguration-flow: todo + repair-issues: todo + stale-devices: todo + + # Platinum + async-dependency: todo + inject-websession: todo + strict-typing: todo diff --git a/custom_components/baidu_voice/strings.json b/custom_components/baidu_voice/strings.json new file mode 100644 index 0000000..d6e3212 --- /dev/null +++ b/custom_components/baidu_voice/strings.json @@ -0,0 +1,21 @@ +{ + "config": { + "step": { + "user": { + "data": { + "host": "[%key:common::config_flow::data::host%]", + "username": "[%key:common::config_flow::data::username%]", + "password": "[%key:common::config_flow::data::password%]" + } + } + }, + "error": { + "cannot_connect": "[%key:common::config_flow::error::cannot_connect%]", + "invalid_auth": "[%key:common::config_flow::error::invalid_auth%]", + "unknown": "[%key:common::config_flow::error::unknown%]" + }, + "abort": { + "already_configured": "[%key:common::config_flow::abort::already_configured_device%]" + } + } +} diff --git a/custom_components/baidu_voice/stt.py b/custom_components/baidu_voice/stt.py new file mode 100644 index 0000000..0ac21e1 --- /dev/null +++ b/custom_components/baidu_voice/stt.py @@ -0,0 +1,147 @@ +"""Support for Baidu speech recognition.""" + +from __future__ import annotations + +import logging +from typing import Any + +from aip import AipSpeech + +from homeassistant.components import stt +from homeassistant.components.stt import ( + AudioBitRates, + AudioChannels, + AudioCodecs, + AudioFormats, + AudioSampleRates, +) +from homeassistant.config_entries import ConfigEntry +from homeassistant.core import HomeAssistant +from homeassistant.helpers.entity_platform import AddConfigEntryEntitiesCallback + +from .const import ( + CONF_API_KEY, + CONF_APP_ID, + CONF_SECRET_KEY, + STT_DEFAULT_LANGUAGE, + STT_LANGUAGES_CODE_MAP, +) + +_LOGGER = logging.getLogger(__name__) + + +async def async_setup_entry( + hass: HomeAssistant, + config_entry: ConfigEntry, + async_add_entities: AddConfigEntryEntitiesCallback, +) -> None: + """Set up Baidu STT platform via config entry.""" + async_add_entities( + [ + BaiduSTTEntity(hass, config_entry.data), + ] + ) + + +class BaiduSTTEntity(stt.SpeechToTextEntity): + """Baidu speech-to-text entity.""" + + def __init__(self, hass: HomeAssistant, config: dict[str, Any]) -> None: + """Initialize Baidu speech-to-text entity.""" + super().__init__() + self.hass = hass + self._config = config + self._client = AipSpeech( + config[CONF_APP_ID], + config[CONF_API_KEY], + config[CONF_SECRET_KEY], + ) + self._attr_name = "Baidu STT" + self._attr_unique_id = f"baidu_stt_{config[CONF_APP_ID]}" + + @property + def supported_languages(self) -> list[str]: + """Return list of supported languages.""" + return ["zh-CN", "en-US", "yue-CN", "sichuan-CN"] + + @property + def supported_formats(self) -> list[str]: + """Return list of supported formats.""" + return [AudioFormats.WAV, "pcm"] + + @property + def supported_codecs(self) -> list[str]: + """Return list of supported codecs.""" + return [AudioCodecs.PCM] + + @property + def supported_bit_rates(self) -> list[int]: + """Return list of supported bit rates.""" + return [AudioBitRates.BITRATE_8, AudioBitRates.BITRATE_16] + + @property + def supported_sample_rates(self) -> list[int]: + """Return list of supported sample rates.""" + return [AudioSampleRates.SAMPLERATE_8000, AudioSampleRates.SAMPLERATE_16000] + + @property + def supported_channels(self) -> list[int]: + """Return list of supported channels.""" + return [AudioChannels.CHANNEL_MONO] + + async def async_process_audio_stream( + self, metadata: stt.SpeechMetadata, stream: stt.AudioStream + ) -> stt.SpeechResult: + """Process an audio stream for speech recognition.""" + try: + audio_data = b"" + async for chunk in stream: + audio_data += chunk + _LOGGER.debug("Metadata: %s", metadata) + result = await self.hass.async_add_executor_job( + self._client.asr, + audio_data, + metadata.format, + metadata.sample_rate, + { + "dev_pid": STT_LANGUAGES_CODE_MAP.get( + metadata.language, STT_DEFAULT_LANGUAGE + ), + "channel": metadata.channel, + }, + ) + + if not isinstance(result, dict): + return stt.SpeechResult( + text=None, + result=stt.SpeechResultState.ERROR, + ) + + if "err_no" in result and result["err_no"] != 0: + _LOGGER.debug( + "Error from Baidu API: %s - %s", + result.get("err_msg", "Unknown error"), + result.get("err_detail", "No details"), + ) + return stt.SpeechResult( + text=None, + result=stt.SpeechResultState.ERROR, + ) + + if "result" not in result or not result["result"]: + return stt.SpeechResult( + text=None, + result=stt.SpeechResultState.NO_SPEECH_DETECTED, + ) + + return stt.SpeechResult( + text=result["result"][0], + result=stt.SpeechResultState.SUCCESS, + ) + + except Exception: # pylint: disable=broad-except + _LOGGER.exception("Error processing Baidu STT") + return stt.SpeechResult( + text=None, + result=stt.SpeechResultState.ERROR, + ) diff --git a/custom_components/baidu_voice/translations/en.json b/custom_components/baidu_voice/translations/en.json new file mode 100644 index 0000000..e3ffda3 --- /dev/null +++ b/custom_components/baidu_voice/translations/en.json @@ -0,0 +1,58 @@ +{ + "config": { + "abort": { + "already_configured": "Service is already configured" + }, + "error": { + "cannot_connect": "Failed to connect", + "invalid_auth": "Invalid authentication", + "unknown": "Unexpected error", + "auth": "Authentication failed", + "test_success": "Connection test successful" + }, + "step": { + "user": { + "title": "Setup Baidu Voice", + "description": "Please enter your Baidu Voice API credentials and TTS settings", + "data": { + "app_id": "App ID", + "api_key": "API Key", + "secret_key": "Secret Key", + "language": "TTS Language", + "stt_language": "STT Language", + "volume": "Volume (0-15)", + "speed": "Speed (0-9)", + "pitch": "Pitch (0-9)", + "voice": "Voice", + "fileformat": "Audio Format", + "test_connection": "Test Connection" + } + } + } + }, + "options": { + "step": { + "init": { + "title": "Update Baidu Voice Settings", + "description": "Modify Baidu Voice settings", + "data": { + "app_id": "App ID", + "api_key": "API Key", + "secret_key": "Secret Key", + "language": "TTS Language", + "stt_language": "STT Language", + "volume": "Volume (0-15)", + "speed": "Speed (0-9)", + "pitch": "Pitch (0-9)", + "voice": "Voice", + "fileformat": "Audio Format", + "test_connection": "Test Connection" + } + } + }, + "error": { + "auth": "Authentication failed", + "test_success": "Connection test successful" + } + } +} \ No newline at end of file diff --git a/custom_components/baidu_voice/translations/zh-Hans.json b/custom_components/baidu_voice/translations/zh-Hans.json new file mode 100644 index 0000000..96985fb --- /dev/null +++ b/custom_components/baidu_voice/translations/zh-Hans.json @@ -0,0 +1,58 @@ +{ + "config": { + "abort": { + "already_configured": "服务已经配置" + }, + "error": { + "cannot_connect": "连接失败", + "invalid_auth": "验证无效", + "unknown": "未知错误", + "auth": "认证失败", + "test_success": "连接测试成功" + }, + "step": { + "user": { + "title": "设置百度语音", + "description": "请输入百度语音API凭据和TTS设置", + "data": { + "app_id": "应用ID", + "api_key": "API密钥", + "secret_key": "密钥", + "language": "TTS语言", + "stt_language": "STT语言", + "volume": "音量 (0-15)", + "speed": "语速 (0-9)", + "pitch": "音调 (0-9)", + "voice": "发音人", + "fileformat": "音频格式", + "test_connection": "测试连接" + } + } + } + }, + "options": { + "step": { + "init": { + "title": "更新百度语音设置", + "description": "修改百度语音设置", + "data": { + "app_id": "应用ID", + "api_key": "API密钥", + "secret_key": "密钥", + "language": "TTS语言", + "stt_language": "STT语言", + "volume": "音量 (0-15)", + "speed": "语速 (0-9)", + "pitch": "音调 (0-9)", + "voice": "发音人", + "fileformat": "音频格式", + "test_connection": "测试连接" + } + } + }, + "error": { + "auth": "认证失败", + "test_success": "连接测试成功" + } + } +} \ No newline at end of file diff --git a/custom_components/baidu_voice/tts.py b/custom_components/baidu_voice/tts.py new file mode 100644 index 0000000..3c97f82 --- /dev/null +++ b/custom_components/baidu_voice/tts.py @@ -0,0 +1,249 @@ +"""Support for Baidu TTS.""" + +from __future__ import annotations + +import logging +from typing import Any + +from aip import AipSpeech + +from homeassistant.components.tts import ( + TextToSpeechEntity, + TtsAudioType, + Voice, + callback, +) +from homeassistant.config_entries import ConfigEntry +from homeassistant.core import HomeAssistant +from homeassistant.exceptions import HomeAssistantError +from homeassistant.helpers.device_registry import DeviceEntryType, DeviceInfo + +from .const import ( + CONF_API_KEY, + CONF_APP_ID, + CONF_SECRET_KEY, + DOMAIN, + TTS_DEFAULT_FILEFORMAT, + TTS_DEFAULT_LANGUAGE, + TTS_DEFAULT_PITCH, + TTS_DEFAULT_SPEED, + TTS_DEFAULT_VOICE, + TTS_DEFAULT_VOLUME, + TTS_FILEFORMAT_MAP, + TTS_LANGUAGES, + TTS_SUPPORTED_VOICES, +) + +_LOGGER = logging.getLogger(__name__) + + +async def async_setup_entry( + hass: HomeAssistant, + config_entry: ConfigEntry, + async_add_entities, +) -> None: + """Set up Baidu TTS from a config entry.""" + _LOGGER.debug("Setting up Baidu TTS") + entity = BaiduTTSEntity(hass, config_entry) + async_add_entities([entity]) + + +class BaiduTTSEntity(TextToSpeechEntity): + """Represent a Baidu TTS entity.""" + + def __init__(self, hass: HomeAssistant, config_entry: ConfigEntry) -> None: + """Initialize the Baidu TTS entity.""" + self.hass = hass + self._config_entry = config_entry + + # Generate unique ID and set name + app_id = config_entry.data[CONF_APP_ID] + self._attr_unique_id = f"baidu_tts_{app_id}" + self._attr_name = "百度语音合成" + + # Add device info + self._attr_device_info = DeviceInfo( + identifiers={(DOMAIN, app_id)}, + name="百度语音", + manufacturer="百度", + model="语音合成服务", + entry_type=DeviceEntryType.SERVICE, + ) + + _LOGGER.debug( + "Initializing Baidu TTS entity with app_id: %s, unique_id: %s, name: %s", + app_id, + self._attr_unique_id, + self._attr_name, + ) + + # Initialize the TTS client + self._client = AipSpeech( + app_id, + config_entry.data[CONF_API_KEY], + config_entry.data[CONF_SECRET_KEY], + ) + + @property + def default_language(self) -> str: + """Return the default language.""" + _LOGGER.debug("Getting default language: %s", TTS_DEFAULT_LANGUAGE) + return TTS_DEFAULT_LANGUAGE + + @property + def supported_languages(self) -> list[str]: + """Return a list of supported languages.""" + languages = list(TTS_LANGUAGES.keys()) + _LOGGER.debug("Getting supported languages: %s", languages) + return languages + + @property + def supported_options(self) -> list[str]: + """Return list of supported options.""" + options = ["speed", "pitch", "volume", "voice", "fileformat"] + _LOGGER.debug("Getting supported options: %s", options) + return options + + @property + def default_options(self) -> dict[str, Any]: + """Return a dict of default options.""" + # Get current values from config_entry.options first, then data, then fall back to defaults + options = { + "speed": self._config_entry.options.get( + "speed", self._config_entry.data.get("speed", TTS_DEFAULT_SPEED) + ), + "pitch": self._config_entry.options.get( + "pitch", self._config_entry.data.get("pitch", TTS_DEFAULT_PITCH) + ), + "volume": self._config_entry.options.get( + "volume", self._config_entry.data.get("volume", TTS_DEFAULT_VOLUME) + ), + "voice": self._config_entry.options.get( + "voice", self._config_entry.data.get("voice", TTS_DEFAULT_VOICE) + ), + "fileformat": self._config_entry.options.get( + "fileformat", + self._config_entry.data.get("fileformat", TTS_DEFAULT_FILEFORMAT), + ), + } + _LOGGER.debug("Getting default options: %s", options) + return options + + @callback + def async_get_supported_voices(self, language: str) -> list[Voice] | None: + """Return a list of supported voices for a language.""" + voices = [] + for voice_id, voice_name in TTS_SUPPORTED_VOICES.items(): + voices.append(Voice(str(voice_id), voice_name)) + _LOGGER.debug("Getting supported voices for language %s: %s", language, voices) + return voices + + def get_tts_audio( + self, message: str, language: str, options: dict[str, Any] + ) -> TtsAudioType: + """Get TTS audio from Baidu.""" + _LOGGER.debug("=================== TTS Debug Info Start ===================") + _LOGGER.debug("Message: %s", message) + _LOGGER.debug("Language: %s", language) + _LOGGER.debug("Raw options dict: %s", options) + _LOGGER.debug("Config entry data: %s", self._config_entry.data) + _LOGGER.debug("Config entry options: %s", self._config_entry.options) + + # 获取配置值,优先级:options > config_entry.options > config_entry.data > 默认值 + try: + speed = int( + options.get("speed") + or self._config_entry.options.get("speed") + or self._config_entry.data.get("speed") + or TTS_DEFAULT_SPEED + ) + pitch = int( + options.get("pitch") + or self._config_entry.options.get("pitch") + or self._config_entry.data.get("pitch") + or TTS_DEFAULT_PITCH + ) + volume = int( + options.get("volume") + or self._config_entry.options.get("volume") + or self._config_entry.data.get("volume") + or TTS_DEFAULT_VOLUME + ) + voice = int( + options.get("voice") + or self._config_entry.options.get("voice") + or self._config_entry.data.get("voice") + or TTS_DEFAULT_VOICE + ) + fileformat = int( + options.get("fileformat") + or self._config_entry.options.get("fileformat") + or self._config_entry.data.get("fileformat") + or TTS_DEFAULT_FILEFORMAT + ) + + _LOGGER.debug( + "Parsed values - speed: %s(%s), pitch: %s(%s), volume: %s(%s), " + "voice: %s(%s), fileformat: %s(%s)", + speed, + type(speed), + pitch, + type(pitch), + volume, + type(volume), + voice, + type(voice), + fileformat, + type(fileformat), + ) + + except ValueError as ex: + _LOGGER.error("Error parsing options: %s", ex) + raise HomeAssistantError(f"Invalid option value: {ex}") from ex + + # 获取音频格式 + format_config = TTS_FILEFORMAT_MAP.get(fileformat, TTS_FILEFORMAT_MAP[6]) + _LOGGER.debug("Using audio format: %s", format_config) + + # 准备API参数 + api_params = { + "spd": speed, + "pit": pitch, + "vol": volume, + "per": voice, + "aue": fileformat, + } + _LOGGER.debug("API parameters: %s", api_params) + + try: + _LOGGER.debug("Calling Baidu TTS API") + result = self._client.synthesis( + message, + language, + 1, # Use standard voice synthesis + api_params, + ) + + if isinstance(result, dict): + error_msg = result.get("err_msg", "Unknown error") + error_no = result.get("err_no", "Unknown") + _LOGGER.error( + "Baidu TTS API error - error_no: %s, error_msg: %s, full_result: %s", + error_no, + error_msg, + result, + ) + return None, None + + _LOGGER.debug("Successfully generated audio, size: %d bytes", len(result)) + + except Exception as ex: + _LOGGER.error( + "Error during TTS generation: %s, type: %s", + str(ex), + type(ex).__name__, + ) + raise HomeAssistantError("Failed to generate TTS audio") from ex + + _LOGGER.debug("=================== TTS Debug Info End ===================") + return format_config, result diff --git a/hacs.json b/hacs.json new file mode 100644 index 0000000..d3bf683 --- /dev/null +++ b/hacs.json @@ -0,0 +1,6 @@ +{ + "name": "Baidu Voice", + "render_readme": true, + "domains": ["tts", "stt"], + "homeassistant": "2023.8.0" + } \ No newline at end of file diff --git a/manifest.json b/manifest.json deleted file mode 100644 index 495e785..0000000 --- a/manifest.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "domain": "baidu_voice", - "name": "Baidu Voice", - "documentation": "https://github.com/howelljiang/baidu-voice", - "dependencies": [], - "codeowners": ["@howelljiang"], - "requirements": ["baidu-aip>=4.16.10", "chardet>=5.2.0"], - "version": "0.1.0", - "iot_class": "cloud_push", - "config_flow": true, - "platforms": ["tts", "stt"] -} \ No newline at end of file diff --git a/settings.png b/settings.png new file mode 100644 index 0000000..323c45a Binary files /dev/null and b/settings.png differ diff --git a/stt.py b/stt.py deleted file mode 100644 index 3166f63..0000000 --- a/stt.py +++ /dev/null @@ -1,120 +0,0 @@ -"""Baidu Speech-to-Text integration.""" -import logging -from typing import Any - -from aip import AipSpeech -from homeassistant.components.stt import ( - AudioBitRates, - AudioChannels, - AudioCodecs, - AudioFormats, - AudioSampleRates, - SpeechToTextEntity, - SpeechMetadata, - SpeechResult, - SpeechResultState, -) -from homeassistant.config_entries import ConfigEntry -from homeassistant.core import HomeAssistant -from homeassistant.helpers.entity_platform import AddEntitiesCallback - -from .const import DOMAIN - -_LOGGER = logging.getLogger(__name__) - -async def async_setup_entry( - hass: HomeAssistant, - config_entry: ConfigEntry, - async_add_entities: AddEntitiesCallback, -) -> bool: - """Set up Baidu STT from a config entry.""" - async_add_entities([BaiduSTTEntity(hass, config_entry)]) - return True - -class BaiduSTTEntity(SpeechToTextEntity): - """Baidu Speech-to-Text entity.""" - - def __init__(self, hass: HomeAssistant, config_entry: ConfigEntry) -> None: - """Initialize Baidu STT entity.""" - self.hass = hass - self._config_entry = config_entry - self._attr_name = "Baidu STT" - self._attr_unique_id = f"{DOMAIN}_stt" - - # 初始化百度语音识别客户端 - self._client = AipSpeech( - config_entry.data["app_id"], - config_entry.data["api_key"], - config_entry.data["secret_key"] - ) - - @property - def supported_languages(self) -> list[str]: - """Return a list of supported languages.""" - return ["zh-CN"] - - @property - def supported_formats(self) -> list[AudioFormats]: - """Return a list of supported formats.""" - return [AudioFormats.WAV, AudioFormats.OGG] - - @property - def supported_codecs(self) -> list[AudioCodecs]: - """Return a list of supported codecs.""" - return [AudioCodecs.PCM, AudioCodecs.OPUS] - - @property - def supported_bit_rates(self) -> list[AudioBitRates]: - """Return a list of supported bit rates.""" - return [AudioBitRates.BITRATE_16] - - @property - def supported_sample_rates(self) -> list[AudioSampleRates]: - """Return a list of supported sample rates.""" - return [AudioSampleRates.SAMPLERATE_16000] - - @property - def supported_channels(self) -> list[AudioChannels]: - """Return a list of supported channels.""" - return [AudioChannels.CHANNEL_MONO] - - async def async_process_audio_stream( - self, metadata: SpeechMetadata, stream: Any - ) -> SpeechResult: - """Process an audio stream to STT service.""" - try: - # 读取音频数据 - audio_data = b"" - async for chunk in stream: - audio_data += chunk - - if not audio_data: - _LOGGER.warning("No audio data received") - return SpeechResult("", SpeechResultState.ERROR) - - _LOGGER.debug("Processing audio stream with metadata: %s", metadata) - _LOGGER.debug("Total audio data size: %d bytes", len(audio_data)) - - # 使用百度语音识别API - result = await self.hass.async_add_executor_job( - self._client.asr, - audio_data, - 'pcm', - 16000, - { - 'dev_pid': 1537, # 普通话(支持简单的英文识别) - } - ) - - _LOGGER.debug("Baidu STT response: %s", result) - - if result.get('err_no') == 0: - text = result.get('result', [""])[0] - return SpeechResult(text, SpeechResultState.SUCCESS) - else: - _LOGGER.error("Baidu STT error: %s", result.get('err_msg')) - return SpeechResult("", SpeechResultState.ERROR) - - except Exception as ex: - _LOGGER.error("Error during Baidu STT: %s", ex, exc_info=True) - return SpeechResult("", SpeechResultState.ERROR) \ No newline at end of file diff --git a/test2audio.pcm b/test2audio.pcm deleted file mode 100644 index 15e6e7b..0000000 Binary files a/test2audio.pcm and /dev/null differ diff --git a/test_baidu_api.py b/test_baidu_api.py deleted file mode 100644 index 06992e2..0000000 --- a/test_baidu_api.py +++ /dev/null @@ -1,111 +0,0 @@ -"""测试百度语音API.""" -import asyncio -import logging -from aip import AipSpeech - -# 配置日志 -logging.basicConfig(level=logging.DEBUG) -_LOGGER = logging.getLogger(__name__) - -# 百度API配置 -APP_ID = "118461597" -API_KEY = "BZk7p3B0DurJuRLrN1cx9iHF" -SECRET_KEY = "sn0i6jtJNPcGI2V13P3MKiGCQYoz9xzf" - -# 测试音频文件路径 -TEST_AUDIO_PATH = "text2audio.pcm" # 修正文件名 - -async def test_tts(): - """测试语音合成.""" - try: - client = AipSpeech(APP_ID, API_KEY, SECRET_KEY) - - # 测试TTS - result = await asyncio.get_event_loop().run_in_executor( - None, - client.synthesis, - "测试语音合成", - "zh", - 1, # 1表示mp3格式 - { - "vol": 5, # 音量,取值0-15,默认为5中音量 - "per": 0, # 发音人选择,0为女声,1为男声,3为情感合成-度逍遥,4为情感合成-度丫丫 - "spd": 5, # 语速,取值0-9,默认为5中语速 - "pit": 5, # 音调,取值0-9,默认为5中语调 - } - ) - - if isinstance(result, dict): - _LOGGER.error("TTS测试失败: %s", result.get('err_msg')) - return False - - # 保存音频文件 - with open('tts_test.mp3', 'wb') as f: - f.write(result) - - _LOGGER.info("TTS测试成功") - return True - - except Exception as ex: - _LOGGER.error("TTS测试异常: %s", ex) - return False - -async def test_stt(): - """测试语音识别.""" - try: - client = AipSpeech(APP_ID, API_KEY, SECRET_KEY) - - # 读取测试音频 - with open(TEST_AUDIO_PATH, 'rb') as f: - audio_data = f.read() - - # 测试STT - result = await asyncio.get_event_loop().run_in_executor( - None, - client.asr, - audio_data, - "pcm", # PCM格式 - 16000, # 采样率 - { - "dev_pid": 1537, # 普通话(支持简单的英文识别) - } - ) - - if result.get("err_no") != 0: - _LOGGER.error("STT测试失败: %s", result.get('err_msg')) - return False - - # 验证识别结果 - recognized_text = result.get("result", [""])[0] - _LOGGER.info("STT识别结果: %s", recognized_text) - - if not recognized_text or "测试" not in recognized_text: - _LOGGER.error("STT识别结果不准确: %s", recognized_text) - return False - - _LOGGER.info("STT测试成功") - return True - - except Exception as ex: - _LOGGER.error("STT测试异常: %s", ex) - return False - -async def main(): - """主测试函数.""" - _LOGGER.info("开始测试百度语音API...") - - # 测试TTS - tts_result = await test_tts() - _LOGGER.info("TTS测试结果: %s", "成功" if tts_result else "失败") - - # 等待1秒 - await asyncio.sleep(1) - - # 测试STT - stt_result = await test_stt() - _LOGGER.info("STT测试结果: %s", "成功" if stt_result else "失败") - - _LOGGER.info("测试完成") - -if __name__ == "__main__": - asyncio.run(main()) \ No newline at end of file diff --git a/text2audio.pcm b/text2audio.pcm deleted file mode 100644 index 15e6e7b..0000000 Binary files a/text2audio.pcm and /dev/null differ diff --git a/tts.py b/tts.py deleted file mode 100644 index 119e24a..0000000 --- a/tts.py +++ /dev/null @@ -1,93 +0,0 @@ -"""Support for Baidu TTS.""" -import logging -from typing import Any - -from aip import AipSpeech -from homeassistant.components.tts import ( - TextToSpeechEntity, - PLATFORM_SCHEMA, -) -from homeassistant.config_entries import ConfigEntry -from homeassistant.core import HomeAssistant -from homeassistant.helpers.entity_platform import AddEntitiesCallback - -from .const import ( - DOMAIN, - DEFAULT_SPEED, - DEFAULT_PITCH, - DEFAULT_VOLUME, - DEFAULT_PERSON, -) - -_LOGGER = logging.getLogger(__name__) - -SUPPORT_LANGUAGES = ["zh", "en"] -DEFAULT_LANG = "zh" - -async def async_setup_entry( - hass: HomeAssistant, - config_entry: ConfigEntry, - async_add_entities: AddEntitiesCallback, -) -> bool: - """Set up Baidu TTS from a config entry.""" - async_add_entities([BaiduTTSEntity(hass, config_entry)]) - return True - -class BaiduTTSEntity(TextToSpeechEntity): - """Baidu TTS Entity.""" - - def __init__(self, hass: HomeAssistant, config_entry: ConfigEntry) -> None: - """Initialize Baidu TTS entity.""" - self.hass = hass - self._config_entry = config_entry - self._attr_name = "Baidu TTS" - self._attr_unique_id = f"{DOMAIN}_tts" - - # 初始化百度语音合成客户端 - self._client = AipSpeech( - config_entry.data["app_id"], - config_entry.data["api_key"], - config_entry.data["secret_key"] - ) - - @property - def default_language(self) -> str: - """Return the default language.""" - return DEFAULT_LANG - - @property - def supported_languages(self) -> list[str]: - """Return list of supported languages.""" - return SUPPORT_LANGUAGES - - async def async_get_tts_audio(self, message: str, language: str, options: dict = None) -> tuple[str, bytes]: - """Load TTS from Baidu.""" - options = options or {} - speed = options.get("speed", DEFAULT_SPEED) - pitch = options.get("pitch", DEFAULT_PITCH) - volume = options.get("volume", DEFAULT_VOLUME) - person = options.get("person", DEFAULT_PERSON) - - try: - result = await self.hass.async_add_executor_job( - self._client.synthesis, - message, - language, - 1, # 1: mp3 - { - "spd": speed, - "pit": pitch, - "vol": volume, - "per": person, - } - ) - - if isinstance(result, dict): - _LOGGER.error("Baidu TTS error: %s", result) - return None, None - - return "mp3", result - - except Exception as ex: - _LOGGER.error("Error during Baidu TTS: %s", ex) - return None, None \ No newline at end of file diff --git a/tts_test.mp3 b/tts_test.mp3 deleted file mode 100644 index fb66419..0000000 Binary files a/tts_test.mp3 and /dev/null differ