diff --git a/app.py b/app.py index 2e1dbeb0..befb5808 100644 --- a/app.py +++ b/app.py @@ -1,8 +1,8 @@ import asyncio from config.logger import setup_logging from config.settings import load_config -from core.server import WebSocketServer -from core.http_server import ConfigServer +from core.websocket_server import WebSocketServer +from manager.http_server import ConfigServer async def main(): setup_logging() # 最先初始化日志 diff --git a/config.yaml b/config.yaml index c0558509..a7644adf 100644 --- a/config.yaml +++ b/config.yaml @@ -1,3 +1,7 @@ +# 如果您是一名开发者,建议阅读以下内容。如果不是开发者,可以忽略这部分内容。 +# 在开发中,建议将【config.yaml】复制一份,改成【.config.yaml】。 系统会优先读取【.config.yaml】文件的配置。 +# 这样做,可以避免在提交代码的时候,错误地提交密钥信息,保护您的密钥安全。 + # 服务器基础配置(Basic server configuration) server: # 服务器监听地址和端口(Server listening address and port) @@ -140,3 +144,31 @@ TTS: output_file: tmp/ access_token: 你的硅基流动API密钥 response_format: wav + FishSpeech: + # 定义TTS API类型 + #启动tts方法: + #python -m tools.api_server + #--listen 0.0.0.0:8080 + #--llama-checkpoint-path "checkpoints/fish-speech-1.5" + #--decoder-checkpoint-path "checkpoints/fish-speech-1.5/firefly-gan-vq-fsq-8x1024-21hz-generator.pth" + #--decoder-config-name firefly_gan_vq + #--compile + type: fishspeech + output_file: tmp/ + response_format: wav + reference_id: null + reference_audio: ["/tmp/test.wav",] + reference_text: ["你弄来这些吟词宴曲来看,还是这些混话来欺负我。",] + normalize: true + max_new_tokens: 1024 + chunk_length: 200 + top_p: 0.7 + repetition_penalty: 1.2 + temperature: 0.7 + streaming: false + use_memory_cache: "on" + seed: null + channels: 1 + rate: 44100 + api_key: "YOUR_API_KEY" + api_url: "http://127.0.0.1:8080/v1/tts" \ No newline at end of file diff --git a/config/settings.py b/config/settings.py index 9bdab1fe..d1f5a984 100644 --- a/config/settings.py +++ b/config/settings.py @@ -1,15 +1,29 @@ import os import argparse +from ruamel.yaml import YAML from core.utils.util import read_config, get_project_dir +def get_config_file(): + default_config_file = "config.yaml" + # 判断是否存在私有的配置文件 + if os.path.exists(get_project_dir() + "." + default_config_file): + default_config_file = "." + default_config_file + return default_config_file + + def load_config(): """加载配置文件""" parser = argparse.ArgumentParser(description="Server configuration") - default_config_file = "config.yaml" - # 判断是否存在私有的配置文件 - if os.path.exists(get_project_dir() + "." + default_config_file): - default_config_file = "." + default_config_file + default_config_file = get_config_file() parser.add_argument("--config_path", type=str, default=default_config_file) args = parser.parse_args() return read_config(args.config_path) + + +def update_config(config): + yaml = YAML() + yaml.preserve_quotes = True + """将配置保存到YAML文件""" + with open(get_config_file(), 'w') as f: + yaml.dump(config, f) diff --git a/core/connection.py b/core/connection.py index 5ba5b819..6b86e2ac 100644 --- a/core/connection.py +++ b/core/connection.py @@ -15,8 +15,8 @@ from core.handle.textHandle import handleTextMessage from core.utils.util import get_string_no_punctuation_or_emoji from concurrent.futures import ThreadPoolExecutor, TimeoutError from core.handle.audioHandle import handleAudioMessage, sendAudioMessage -from .auth import AuthMiddleware, AuthenticationError -from config.private_config import PrivateConfig # Updated import path +from config.private_config import PrivateConfig +from core.auth import AuthMiddleware, AuthenticationError class ConnectionHandler: def __init__(self, config: Dict[str, Any], _vad, _asr, _llm, _tts): diff --git a/core/providers/tts/fishspeech.py b/core/providers/tts/fishspeech.py new file mode 100644 index 00000000..c7121498 --- /dev/null +++ b/core/providers/tts/fishspeech.py @@ -0,0 +1,152 @@ + +import base64 +import os +import uuid +import requests +import ormsgpack +from pathlib import Path +from pydantic import BaseModel, Field, conint, model_validator +from typing_extensions import Annotated +from datetime import datetime +from typing import Literal +# from base import TTSProviderBase +from core.providers.tts.base import TTSProviderBase + + +class ServeReferenceAudio(BaseModel): + audio: bytes + text: str + + @model_validator(mode="before") + def decode_audio(cls, values): + audio = values.get("audio") + if ( + isinstance(audio, str) and len(audio) > 255 + ): # Check if audio is a string (Base64) + try: + values["audio"] = base64.b64decode(audio) + except Exception as e: + # If the audio is not a valid base64 string, we will just ignore it and let the server handle it + pass + return values + + def __repr__(self) -> str: + return f"ServeReferenceAudio(text={self.text!r}, audio_size={len(self.audio)})" + +class ServeTTSRequest(BaseModel): + text: str + chunk_length: Annotated[int, conint(ge=100, le=300, strict=True)] = 200 + # Audio format + format: Literal["wav", "pcm", "mp3"] = "wav" + # References audios for in-context learning + references: list[ServeReferenceAudio] = [] + # Reference id + # For example, if you want use https://fish.audio/m/7f92f8afb8ec43bf81429cc1c9199cb1/ + # Just pass 7f92f8afb8ec43bf81429cc1c9199cb1 + reference_id: str | None = None + seed: int | None = None + use_memory_cache: Literal["on", "off"] = "off" + # Normalize text for en & zh, this increase stability for numbers + normalize: bool = True + # not usually used below + streaming: bool = False + max_new_tokens: int = 1024 + top_p: Annotated[float, Field(ge=0.1, le=1.0, strict=True)] = 0.7 + repetition_penalty: Annotated[float, Field(ge=0.9, le=2.0, strict=True)] = 1.2 + temperature: Annotated[float, Field(ge=0.1, le=1.0, strict=True)] = 0.7 + + class Config: + # Allow arbitrary types for pytorch related types + arbitrary_types_allowed = True + + +def audio_to_bytes(file_path): + if not file_path or not Path(file_path).exists(): + return None + with open(file_path, "rb") as wav_file: + wav = wav_file.read() + return wav + +def read_ref_text(ref_text): + path = Path(ref_text) + if path.exists() and path.is_file(): + with path.open("r", encoding="utf-8") as file: + return file.read() + return ref_text + +class TTSProvider(TTSProviderBase): + + def __init__(self, config, delete_audio_file): + super().__init__(config, delete_audio_file) + + self.reference_id = config.get("reference_id") + self.reference_audio = config.get("reference_audio",[]) + self.reference_text = config.get("reference_text",[]) + self.format = config.get("format","wav") + self.channels = config.get("channels",1) + self.rate = config.get("rate",44100) + self.api_key = config.get("api_key","YOUR_API_KEY") + self.normalize = config.get("normalize",True) + self.max_new_tokens = config.get("max_new_tokens",1024) + self.chunk_length = config.get("chunk_length",200) + self.top_p = config.get("top_p",0.7) + self.repetition_penalty = config.get("repetition_penalty",1.2) + self.temperature = config.get("temperature",0.7) + self.streaming = config.get("streaming",False) + self.use_memory_cache = config.get("use_memory_cache","on") + self.seed = config.get("seed") + self.api_url = config.get("api_url","http://127.0.0.1:8080/v1/tts") + + def generate_filename(self, extension=".wav"): + return os.path.join(self.output_file, f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}") + + async def text_to_speak(self, text, output_file): + # Prepare reference data + byte_audios = [audio_to_bytes(ref_audio) for ref_audio in self.reference_audio] + ref_texts = [read_ref_text(ref_text) for ref_text in self.reference_text] + + data = { + "text": text, + "references": [ + ServeReferenceAudio( + audio=audio if audio else b"", text=text + ) + for text, audio in zip(ref_texts, byte_audios) + ], + "reference_id": self.reference_id, + "normalize": self.normalize, + "format": self.format, + "max_new_tokens": self.max_new_tokens, + "chunk_length": self.chunk_length, + "top_p": self.top_p, + "repetition_penalty": self.repetition_penalty, + "temperature": self.temperature, + "streaming": self.streaming, + "use_memory_cache": self.use_memory_cache, + "seed": self.seed, + } + + pydantic_data = ServeTTSRequest(**data) + + response = requests.post( + self.api_url, + data=ormsgpack.packb(pydantic_data, option=ormsgpack.OPT_SERIALIZE_PYDANTIC), + headers={ + "Authorization": f"Bearer {self.api_key}", + "Content-Type": "application/msgpack", + }, + ) + + if response.status_code == 200: + audio_content = response.content + + with open(output_file, "wb") as audio_file: + audio_file.write(audio_content) + + + + else: + print(f"Request failed with status code {response.status_code}") + print(response.json()) + + diff --git a/core/server.py b/core/websocket_server.py similarity index 100% rename from core/server.py rename to core/websocket_server.py diff --git a/manager/api/prompt.py b/manager/api/prompt.py index a9ce8238..1fe5aa08 100644 --- a/manager/api/prompt.py +++ b/manager/api/prompt.py @@ -1,5 +1,7 @@ import logging from aiohttp import web +from config.settings import update_config +from ruamel.yaml.scalarstring import PreservedScalarString from manager.api.auth import verify_token from manager.api.response import response_unauthorized, response_success, response_error @@ -28,10 +30,13 @@ class PromptApi: if 'prompt' not in data: return response_success() - self.config['prompt'] = data['prompt'] - # TODO 保存到配置文件 - return response_success() + # 使用PreservedScalarString保留多行文本格式 + self.config['prompt'] = PreservedScalarString(data['prompt']) + # 保存到配置文件 + update_config(self.config) + + return response_success() except Exception as e: logger.error(f"Failed to update prompt: {e}") return response_error(str(e)) diff --git a/core/http_server.py b/manager/http_server.py similarity index 100% rename from core/http_server.py rename to manager/http_server.py diff --git a/manager/static/css/common.css b/manager/static/css/common.css new file mode 100644 index 00000000..b6f7287e --- /dev/null +++ b/manager/static/css/common.css @@ -0,0 +1,77 @@ +/* common.css */ +/* 基础样式 */ +body { + margin: 0; + padding: 0; + background: linear-gradient(135deg, #1a1f25 0%, #0d1117 100%); + min-height: 100vh; + font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Oxygen, Ubuntu, Cantarell, 'Open Sans', 'Helvetica Neue', sans-serif; +} + +/* 头部组件 */ +.app-header { + background: rgba(13, 17, 23, 0.8); + backdrop-filter: blur(10px); + color: #fff; + padding: 1.5rem 2rem; + font-size: 1.4rem; + border-bottom: 1px solid rgba(255, 255, 255, 0.1); + display: flex; + align-items: center; + gap: 1rem; + position: relative; + z-index: 1; +} + +.header-logo { + width: 32px; + height: 32px; + background: linear-gradient(45deg, #00c6fb 0%, #005bea 100%); + border-radius: 8px; + display: flex; + align-items: center; + justify-content: center; + font-weight: bold; +} + +/* 页脚组件 */ +.app-footer { + text-align: center; + color: rgba(255, 255, 255, 0.6); + font-size: 13px; + padding: 20px; + position: fixed; + bottom: 0; + width: 100%; + z-index: 1; +} + +/* 背景动画 */ +.animated-bg { + position: fixed; + top: 0; + left: 0; + width: 100%; + height: 100%; + z-index: -1; + background: linear-gradient(-45deg, #1a1f25, #0d1117, #162030, #1c1c1c); + background-size: 400% 400%; +} + + +.user-menu { + position: fixed; + top: 23px; + right: 32px; + z-index: 1000; + cursor: pointer; +} + +.user-info { + align-items: center; +} + +.user-name { + font-size: 14px; + color: #606266; +} \ No newline at end of file diff --git a/manager/static/images/favicon.ico b/manager/static/images/favicon.ico new file mode 100644 index 00000000..2e4ba64e Binary files /dev/null and b/manager/static/images/favicon.ico differ diff --git a/manager/static/index.html b/manager/static/index.html index c42fcfed..2e087027 100644 --- a/manager/static/index.html +++ b/manager/static/index.html @@ -1,5 +1,5 @@ - - + +
@@ -8,39 +8,11 @@ + +