* 快速分类音频并把yml格式结果存在训练根目录里 (#190) * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> * Update models.py * Update webui.py * Update infer.py * Create compress_model.py * 重新提交,更新Gradio推理UI (#193) * Update webui.py * Update webui.py * 更新 train_ms.py * 更新 models.py * 更新 models.py * 更新 models.py * 更新 train_ms.py * 更新 train_ms.py * 更新 models.py * Update preprocess_text.py * Update config.json * Update train_ms.py * Update webui.py (#206) * Add files via upload (#209) * Update train_ms.py * Update train_ms.py * Update preprocess_text.py * Update train_ms.py * fix (#211) * Update emotion_clustering.py * Add files via upload * Update emotion_clustering.py * add cluster center save * Add files via upload * Update config.py * Update default_config.yml * Update config.py * Update config.py * Update emotion_clustering.py * Update emotion_clustering.py * Update config.py * Update emotion_clustering.py * Update emotion_clustering.py * Update webui.py * Update emotion_clustering.py * Update commons.py * Update emotion_clustering.py * Update webui.py * Update webui.py * Add files via upload * Update train_ms.py * Update train_ms.py * Update train_ms.py * Update train_ms.py * Update train_ms.py * Update webui.py * Update emotion_clustering.py * Update emotion_clustering.py * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * fix default_config.yml. * Update infer.py * feat: support infer 2.1 models * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * fix: support infer 2.1 models 兼容bug修复 * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Update train_ms.py * Add CLAP * Fix data loader * Fix infer.py * Fix webui.py * Add prompt template * Update clap_gen.py * Fix wrong environ value * Add g for dur disc * Update clap_gen.py * Fix multilang generation * Update config.json * Prompt mode * Improve slice segments performance * Add preprocess webui * Update webui_preprocess.py * Update webui_preprocess.py * Update config.py * Update default_config.yml * Update config.py * Update clap_gen.py * Delete emo_gen.py * Delete get_emo.py * Delete emotional/wav2vec2-large-robust-12-ft-emotion-msp-dim directory * Update README.md * Update README * Split val per lang * Delete emotion_clustering.py * Update default_config.yml * Update default_config.yml * Update config.py * Update preprocess_text.py * Update webui_preprocess.py * Update defalut_config.yml * Update webui_preprocess.py * Update preprocess_text.py * Random augmentation for CLAP * Update data_utils.py * Update preprocess_text.py * Add vq for CLAP features to avoid overfitting * Random dummy inputs * Update webui.py * Update models.py * Update infer.py * Apply Code Formatter Change * Update config.json * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: YYuX-1145 <138500330+YYuX-1145@users.noreply.github.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> Co-authored-by: Sora <654163754@qq.com> Co-authored-by: Sihan Wang <wangsihan1995@gmail.com> Co-authored-by: Stardust-minus <Stardust-minus@users.noreply.github.com>
142 lines
5.1 KiB
Python
142 lines
5.1 KiB
Python
import json
|
||
from collections import defaultdict
|
||
from random import shuffle
|
||
from typing import Optional
|
||
import os
|
||
|
||
from tqdm import tqdm
|
||
import click
|
||
from text.cleaner import clean_text
|
||
from config import config
|
||
from infer import latest_version
|
||
|
||
preprocess_text_config = config.preprocess_text_config
|
||
|
||
|
||
@click.command()
|
||
@click.option(
|
||
"--transcription-path",
|
||
default=preprocess_text_config.transcription_path,
|
||
type=click.Path(exists=True, file_okay=True, dir_okay=False),
|
||
)
|
||
@click.option("--cleaned-path", default=preprocess_text_config.cleaned_path)
|
||
@click.option("--train-path", default=preprocess_text_config.train_path)
|
||
@click.option("--val-path", default=preprocess_text_config.val_path)
|
||
@click.option(
|
||
"--config-path",
|
||
default=preprocess_text_config.config_path,
|
||
type=click.Path(exists=True, file_okay=True, dir_okay=False),
|
||
)
|
||
@click.option("--val-per-lang", default=preprocess_text_config.val_per_lang)
|
||
@click.option("--max-val-total", default=preprocess_text_config.max_val_total)
|
||
@click.option("--clean/--no-clean", default=preprocess_text_config.clean)
|
||
@click.option("-y", "--yml_config")
|
||
def preprocess(
|
||
transcription_path: str,
|
||
cleaned_path: Optional[str],
|
||
train_path: str,
|
||
val_path: str,
|
||
config_path: str,
|
||
val_per_lang: int,
|
||
max_val_total: int,
|
||
clean: bool,
|
||
yml_config: str, # 这个不要删
|
||
):
|
||
if cleaned_path == "" or cleaned_path is None:
|
||
cleaned_path = transcription_path + ".cleaned"
|
||
|
||
if clean:
|
||
with open(cleaned_path, "w", encoding="utf-8") as out_file:
|
||
with open(transcription_path, "r", encoding="utf-8") as trans_file:
|
||
lines = trans_file.readlines()
|
||
# print(lines, ' ', len(lines))
|
||
if len(lines) != 0:
|
||
for line in tqdm(lines):
|
||
try:
|
||
utt, spk, language, text = line.strip().split("|")
|
||
norm_text, phones, tones, word2ph = clean_text(
|
||
text, language
|
||
)
|
||
out_file.write(
|
||
"{}|{}|{}|{}|{}|{}|{}\n".format(
|
||
utt,
|
||
spk,
|
||
language,
|
||
norm_text,
|
||
" ".join(phones),
|
||
" ".join([str(i) for i in tones]),
|
||
" ".join([str(i) for i in word2ph]),
|
||
)
|
||
)
|
||
except Exception as e:
|
||
print(line)
|
||
print(f"生成训练集和验证集时发生错误!, 详细信息:\n{e}")
|
||
|
||
transcription_path = cleaned_path
|
||
spk_utt_map = defaultdict(list)
|
||
spk_id_map = {}
|
||
current_sid = 0
|
||
|
||
with open(transcription_path, "r", encoding="utf-8") as f:
|
||
audioPaths = set()
|
||
countSame = 0
|
||
countNotFound = 0
|
||
for line in f.readlines():
|
||
utt, spk, language, text, phones, tones, word2ph = line.strip().split("|")
|
||
if utt in audioPaths:
|
||
# 过滤数据集错误:相同的音频匹配多个文本,导致后续bert出问题
|
||
print(f"重复音频文本:{line}")
|
||
countSame += 1
|
||
continue
|
||
if not os.path.isfile(utt):
|
||
# 过滤数据集错误:不存在对应音频
|
||
print(f"没有找到对应的音频:{utt}")
|
||
countNotFound += 1
|
||
continue
|
||
audioPaths.add(utt)
|
||
spk_utt_map[language].append(line)
|
||
if spk not in spk_id_map.keys():
|
||
spk_id_map[spk] = current_sid
|
||
current_sid += 1
|
||
print(f"总重复音频数:{countSame},总未找到的音频数:{countNotFound}")
|
||
|
||
train_list = []
|
||
val_list = []
|
||
|
||
for spk, utts in spk_utt_map.items():
|
||
shuffle(utts)
|
||
val_list += utts[:val_per_lang]
|
||
train_list += utts[val_per_lang:]
|
||
|
||
shuffle(val_list)
|
||
if len(val_list) > max_val_total:
|
||
train_list += val_list[max_val_total:]
|
||
val_list = val_list[:max_val_total]
|
||
|
||
with open(train_path, "w", encoding="utf-8") as f:
|
||
for line in train_list:
|
||
f.write(line)
|
||
|
||
with open(val_path, "w", encoding="utf-8") as f:
|
||
for line in val_list:
|
||
f.write(line)
|
||
|
||
json_config = json.load(open(config_path, encoding="utf-8"))
|
||
json_config["data"]["spk2id"] = spk_id_map
|
||
json_config["data"]["n_speakers"] = len(spk_id_map)
|
||
# 新增写入:写入训练版本、数据集路径
|
||
json_config["version"] = latest_version
|
||
json_config["data"]["training_files"] = os.path.normpath(train_path).replace(
|
||
"\\", "/"
|
||
)
|
||
json_config["data"]["validation_files"] = os.path.normpath(val_path).replace(
|
||
"\\", "/"
|
||
)
|
||
with open(config_path, "w", encoding="utf-8") as f:
|
||
json.dump(json_config, f, indent=2, ensure_ascii=False)
|
||
print("训练集和验证集生成完成!")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
preprocess()
|