mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-22 23:23:55 +08:00
93 lines
3.0 KiB
Python
93 lines
3.0 KiB
Python
import yaml
|
||||
|
|
import unicodedata
|
|||
|
|
import socket
|
|||
|
|
import os
|
|||
|
|
import json
|
|||
|
|
|
|||
|
|
|
|||
|
|
def get_project_dir():
|
|||
|
|
projectName = 'xiaozhi-esp32-server'
|
|||
|
|
filePath = os.path.abspath(__file__)
|
|||
|
|
return filePath[:filePath.rfind('/' + projectName + '/') + len(projectName) + 2]
|
|||
|
|
|
|||
|
|
|
|||
|
|
def get_local_ip():
|
|||
|
|
try:
|
|||
|
|
s = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
|
|||
|
|
# Connect to Google's DNS servers
|
|||
|
|
s.connect(("8.8.8.8", 80))
|
|||
|
|
local_ip = s.getsockname()[0]
|
|||
|
|
s.close()
|
|||
|
|
return local_ip
|
|||
|
|
except Exception as e:
|
|||
|
|
return "127.0.0.1"
|
|||
|
|
|
|||
|
|
|
|||
|
|
def read_config(config_path):
|
|||
|
|
with open(config_path, "r", encoding="utf-8") as file:
|
|||
|
|
config = yaml.safe_load(file)
|
|||
|
|
return config
|
|||
|
|
|
|||
|
|
|
|||
|
|
def write_json_file(file_path, data):
|
|||
|
|
"""将数据写入 JSON 文件"""
|
|||
|
|
with open(file_path, 'w', encoding='utf-8') as file:
|
|||
|
|
json.dump(data, file, ensure_ascii=False, indent=4)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def is_segment(tokens):
|
|||
|
|
if tokens[-1] in (",", ".", "?", ",", "。", "?", "!", "!", ";", ";", ":", ":"):
|
|||
|
|
return True
|
|||
|
|
else:
|
|||
|
|
return False
|
|||
|
|
|
|||
|
|
def is_punctuation_or_emoji(char):
|
|||
|
|
"""检查字符是否为空格、指定标点或表情符号"""
|
|||
|
|
# 定义需要去除的中英文标点(包括全角/半角)
|
|||
|
|
punctuation_set = {
|
|||
|
|
',', ',', # 中文逗号 + 英文逗号
|
|||
|
|
'。', '.', # 中文句号 + 英文句号
|
|||
|
|
'!', '!', # 中文感叹号 + 英文感叹号
|
|||
|
|
'-', '-', # 英文连字符 + 中文全角横线
|
|||
|
|
'、' # 中文顿号
|
|||
|
|
}
|
|||
|
|
if char.isspace() or char in punctuation_set:
|
|||
|
|
return True
|
|||
|
|
# 检查表情符号(保留原有逻辑)
|
|||
|
|
code_point = ord(char)
|
|||
|
|
emoji_ranges = [
|
|||
|
|
(0x1F600, 0x1F64F), (0x1F300, 0x1F5FF),
|
|||
|
|
(0x1F680, 0x1F6FF), (0x1F900, 0x1F9FF),
|
|||
|
|
(0x1FA70, 0x1FAFF), (0x2600, 0x26FF),
|
|||
|
|
(0x2700, 0x27BF)
|
|||
|
|
]
|
|||
|
|
return any(start <= code_point <= end for start, end in emoji_ranges)
|
|||
|
|
|
|||
|
|
def get_string_no_punctuation_or_emoji(s):
|
|||
|
|
"""去除字符串首尾的空格、标点符号和表情符号"""
|
|||
|
|
chars = list(s)
|
|||
|
|
# 处理开头的字符
|
|||
|
|
start = 0
|
|||
|
|
while start < len(chars) and is_punctuation_or_emoji(chars[start]):
|
|||
|
|
start += 1
|
|||
|
|
# 处理结尾的字符
|
|||
|
|
end = len(chars) - 1
|
|||
|
|
while end >= start and is_punctuation_or_emoji(chars[end]):
|
|||
|
|
end -= 1
|
|||
|
|
return ''.join(chars[start:end+1])
|
|||
|
|
|
|||
|
|
def remove_punctuation_and_length(text):
|
|||
|
|
# 全角符号和半角符号的Unicode范围
|
|||
|
|
full_width_punctuations = '!"#$%&'()*+,-。/:;<=>?@[\]^_`{|}~'
|
|||
|
|
half_width_punctuations = '!"#$%&\'()*+,-./:;<=>?@[\]^_`{|}~'
|
|||
|
|
space = ' ' # 半角空格
|
|||
|
|
full_width_space = ' ' # 全角空格
|
|||
|
|
|
|||
|
|
# 去除全角和半角符号以及空格
|
|||
|
|
result = ''.join([char for char in text if
|
|||
|
|
char not in full_width_punctuations and char not in half_width_punctuations and char not in space and char not in full_width_space])
|
|||
|
|
|
|||
|
|
if result == "Yeah":
|
|||
|
|
return 0
|
|||
|
|
return len(result)
|