* Fix inputs of duration discriminator * Add LSTM * Update models.py * Update tensorboard scalar * Noise injection for minimizing modality gap * Update infer.py * support bf16 run * del unused_para flag * support bf16 config * add grad clip * fix(logger and grad):add dur grad,fix grad clip * Update webui_preprocess.py * Fix English G2P * fix(bert_gen):add pass * Pass SDP to DD * Update webui_preprocess.py * Update config.json * Update webui.py * Update chinese_bert.py * Upload webui for deploy * Update webui.py * torch.save as pt not npy * Update config.json * add freeze emo vq * Update webui_preprocess.py * Fix tone_sandhi.py * Comment up grad clip * Fix in-place addition * Add SLM discriminator * Add DDP for WD * Feat: Style text: make emotions and style similar to the style text by mixing bert (#240) (#241) * fix:(oldVersion210) Load on demand Emotion model * feat: update fastapi.py. 添加更多错误日志信息 * Switch pyopenjtalk to pyopenjtalk-prebuilt * fix: update fastapi.py. 2.2 reference适配 * Update resample.py * 修复Onnx导出的BUG (#237) * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Add files via upload * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Delete attentions_onnx.py * Delete models_onnx.py * Add files via upload * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Update __init__.py * Update __init__.py * Update __init__.py * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- * Fix onnx * Format export * Feat: style-text and bert mixing (JA only) * Ensure the same tensor shape * Update * update gradio version * Fix * Style text for chinese and english (ver 2.2) * Style text for chinese and english (ver 2.1) * Style text in FastAPI * Translate style text desc in chinese --------- Co-authored-by: litagin02 <139731664+litagin02@users.noreply.github.com> Co-authored-by: Sora <654163754@qq.com> Co-authored-by: Sihan Wang <wangsihan1995@gmail.com> Co-authored-by: Ναρουσέ·μ·γιουμεμί·Χινακάννα <40709280+NaruseMioShirakana@users.noreply.github.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> * Remove CLAP * Revert "Remove CLAP" This reverts commit 62fd59bc837c580239840a2bc84b15e0663730fc. Revert * Remove CLAP * bf16 audo grad cilp * Update webui and infer utils * Update webui.py * Update webui.py * Update webui-preprocess.py * Update webui_preprocess.py --------- Co-authored-by: Sihan Wang <wangsihan1995@gmail.com> Co-authored-by: OedoSoldier <31711261+OedoSoldier@users.noreply.github.com> Co-authored-by: litagin02 <139731664+litagin02@users.noreply.github.com> Co-authored-by: Sora <654163754@qq.com> Co-authored-by: Ναρουσέ·μ·γιουμεμί·Χινακάννα <40709280+NaruseMioShirakana@users.noreply.github.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
82 lines
2.8 KiB
Python
82 lines
2.8 KiB
Python
import re
|
||
|
||
|
||
def extract_language_and_text_updated(speaker, dialogue):
|
||
# 使用正则表达式匹配<语言>标签和其后的文本
|
||
pattern_language_text = r"<(\S+?)>([^<]+)"
|
||
matches = re.findall(pattern_language_text, dialogue, re.DOTALL)
|
||
speaker = speaker[1:-1]
|
||
# 清理文本:去除两边的空白字符
|
||
matches_cleaned = [(lang.upper(), text.strip()) for lang, text in matches]
|
||
matches_cleaned.append(speaker)
|
||
return matches_cleaned
|
||
|
||
|
||
def validate_text(input_text):
|
||
# 验证说话人的正则表达式
|
||
pattern_speaker = r"(\[\S+?\])((?:\s*<\S+?>[^<\[\]]+?)+)"
|
||
|
||
# 使用re.DOTALL标志使.匹配包括换行符在内的所有字符
|
||
matches = re.findall(pattern_speaker, input_text, re.DOTALL)
|
||
|
||
# 对每个匹配到的说话人内容进行进一步验证
|
||
for _, dialogue in matches:
|
||
language_text_matches = extract_language_and_text_updated(_, dialogue)
|
||
if not language_text_matches:
|
||
return (
|
||
False,
|
||
"Error: Invalid format detected in dialogue content. Please check your input.",
|
||
)
|
||
|
||
# 如果输入的文本中没有找到任何匹配项
|
||
if not matches:
|
||
return (
|
||
False,
|
||
"Error: No valid speaker format detected. Please check your input.",
|
||
)
|
||
|
||
return True, "Input is valid."
|
||
|
||
|
||
def text_matching(text: str) -> list:
|
||
speaker_pattern = r"(\[\S+?\])(.+?)(?=\[\S+?\]|$)"
|
||
matches = re.findall(speaker_pattern, text, re.DOTALL)
|
||
result = []
|
||
for speaker, dialogue in matches:
|
||
result.append(extract_language_and_text_updated(speaker, dialogue))
|
||
return result
|
||
|
||
|
||
def cut_para(text):
|
||
splitted_para = re.split("[\n]", text) # 按段分
|
||
splitted_para = [
|
||
sentence.strip() for sentence in splitted_para if sentence.strip()
|
||
] # 删除空字符串
|
||
return splitted_para
|
||
|
||
|
||
def cut_sent(para):
|
||
para = re.sub("([。!;?\?])([^”’])", r"\1\n\2", para) # 单字符断句符
|
||
para = re.sub("(\.{6})([^”’])", r"\1\n\2", para) # 英文省略号
|
||
para = re.sub("(\…{2})([^”’])", r"\1\n\2", para) # 中文省略号
|
||
para = re.sub("([。!?\?][”’])([^,。!?\?])", r"\1\n\2", para)
|
||
para = para.rstrip() # 段尾如果有多余的\n就去掉它
|
||
return para.split("\n")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
text = """
|
||
[说话人1]
|
||
[说话人2]<zh>你好吗?<jp>元気ですか?<jp>こんにちは,世界。<zh>你好吗?
|
||
[说话人3]<zh>谢谢。<jp>どういたしまして。
|
||
"""
|
||
text_matching(text)
|
||
# 测试函数
|
||
test_text = """
|
||
[说话人1]<zh>你好,こんにちは!<jp>こんにちは,世界。
|
||
[说话人2]<zh>你好吗?
|
||
"""
|
||
text_matching(test_text)
|
||
res = validate_text(test_text)
|
||
print(res)
|