This commit is contained in:
rcell
2023-07-24 15:16:06 +08:00
parent 1fa3ce1d03
commit 9fd6e9a259
7 changed files with 348 additions and 50345 deletions

View File

@@ -13,7 +13,7 @@
"batch_size": 32,
"fp16_run": true,
"lr_decay": 0.999875,
"segment_size": 8192,
"segment_size": 16384,
"init_lr_ratio": 1,
"warmup_epochs": 0,
"c_mel": 45,
@@ -23,161 +23,194 @@
"training_files": "filelists/train.list",
"validation_files": "filelists/val.list",
"max_wav_value": 32768.0,
"sampling_rate": 22050,
"filter_length": 1024,
"hop_length": 256,
"win_length": 1024,
"n_mel_channels": 80,
"sampling_rate": 44100,
"filter_length": 2048,
"hop_length": 512,
"win_length": 2048,
"n_mel_channels": 128,
"mel_fmin": 0.0,
"mel_fmax": null,
"add_blank": true,
"n_speakers": 300,
"cleaned_text": true,
"spk2id": {
"SSB0080": 0,
"SSB0012": 1,
"SSB0038": 2,
"SSB0382": 3,
"SSB0394": 4,
"SSB0395": 5,
"SSB0316": 6,
"SSB0200": 7,
"SSB1408": 8,
"SSB1392": 9,
"SSB0700": 10,
"SSB1138": 11,
"SSB1072": 12,
"SSB0751": 13,
"SSB0338": 14,
"SSB0435": 15,
"SSB0913": 16,
"SSB1806": 17,
"SSB1878": 18,
"SSB1385": 19,
"SSB0817": 20,
"SSB0599": 21,
"SSB0887": 22,
"SSB0720": 23,
"SSB1091": 24,
"SSB0786": 25,
"SSB0737": 26,
"SSB0666": 27,
"SSB0606": 28,
"SSB0535": 29,
"SSB0112": 30,
"SSB1161": 31,
"SSB1448": 32,
"SSB1684": 33,
"SSB1699": 34,
"SSB1341": 35,
"SSB0919": 36,
"SSB1056": 37,
"SSB1115": 38,
"SSB1563": 39,
"SSB0482": 40,
"SSB0502": 41,
"SSB0415": 42,
"SSB0145": 43,
"0001_Angry": 44,
"0001_Happy": 45,
"0001_Neutral": 46,
"0001_Sad": 47,
"0001_Surprise": 48,
"0002_Angry": 49,
"0002_Happy": 50,
"0002_Neutral": 51,
"0002_Sad": 52,
"0002_Surprise": 53,
"0003_Angry": 54,
"0003_Happy": 55,
"0003_Neutral": 56,
"0003_Sad": 57,
"0003_Surprise": 58,
"0004_Angry": 59,
"0004_Happy": 60,
"0004_Neutral": 61,
"0004_Sad": 62,
"0004_Surprise": 63,
"0005_Angry": 64,
"0005_Happy": 65,
"0005_Neutral": 66,
"0005_Sad": 67,
"0005_Surprise": 68,
"0006_Angry": 69,
"0006_Happy": 70,
"0006_Neutral": 71,
"0006_Sad": 72,
"0006_Surprise": 73,
"0007_Angry": 74,
"0007_Happy": 75,
"0007_Neutral": 76,
"0007_Sad": 77,
"0007_Surprise": 78,
"0008_Angry": 79,
"0008_Happy": 80,
"0008_Neutral": 81,
"0008_Sad": 82,
"0008_Surprise": 83,
"0009_Angry": 84,
"0009_Happy": 85,
"0009_Neutral": 86,
"0009_Sad": 87,
"0009_Surprise": 88,
"0010_Angry": 89,
"0010_Happy": 90,
"0010_Neutral": 91,
"0010_Sad": 92,
"0010_Surprise": 93,
"0011_Angry": 94,
"0011_Happy": 95,
"0011_Neutral": 96,
"0011_Sad": 97,
"0011_Surprise": 98,
"0012_Angry": 99,
"0012_Happy": 100,
"0012_Neutral": 101,
"0012_Sad": 102,
"0012_Surprise": 103,
"0013_Angry": 104,
"0013_Happy": 105,
"0013_Neutral": 106,
"0013_Sad": 107,
"0013_Surprise": 108,
"0014_Angry": 109,
"0014_Happy": 110,
"0014_Neutral": 111,
"0014_Sad": 112,
"0014_Surprise": 113,
"0015_Angry": 114,
"0015_Happy": 115,
"0015_Neutral": 116,
"0015_Sad": 117,
"0015_Surprise": 118,
"0016_Angry": 119,
"0016_Happy": 120,
"0016_Neutral": 121,
"0016_Sad": 122,
"0016_Surprise": 123,
"0017_Angry": 124,
"0017_Happy": 125,
"0017_Neutral": 126,
"0017_Sad": 127,
"0017_Surprise": 128,
"0018_Angry": 129,
"0018_Happy": 130,
"0018_Neutral": 131,
"0018_Sad": 132,
"0018_Surprise": 133,
"0019_Angry": 134,
"0019_Happy": 135,
"0019_Neutral": 136,
"0019_Sad": 137,
"0019_Surprise": 138,
"0020_Angry": 139,
"0020_Happy": 140,
"0020_Neutral": 141,
"0020_Sad": 142,
"0020_Surprise": 143
"": 0,
"": 1,
"派蒙": 2,
"纳西妲": 3,
"阿贝多": 4,
"温迪": 5,
"枫原万叶": 6,
"钟离": 7,
"荒泷一斗": 8,
"八重神子": 9,
"艾尔海森": 10,
"提纳里": 11,
"迪希雅": 12,
"卡维": 13,
"宵宫": 14,
"莱依拉": 15,
"赛诺": 16,
"诺艾尔": 17,
"托马": 18,
"凝光": 19,
"莫娜": 20,
"北斗": 21,
"神里绫华": 22,
"雷电将军": 23,
"芭芭拉": 24,
"鹿野院平藏": 25,
"五郎": 26,
"迪奥娜": 27,
"凯亚": 28,
"安柏": 29,
"班尼特": 30,
"": 31,
"柯莱": 32,
"夜兰": 33,
"妮露": 34,
"辛焱": 35,
"珐露珊": 36,
"": 37,
"香菱": 38,
"达达利亚": 39,
"砂糖": 40,
"早柚": 41,
"云堇": 42,
"刻晴": 43,
"丽莎": 44,
"迪卢克": 45,
"烟绯": 46,
"重云": 47,
"珊瑚宫心海": 48,
"胡桃": 49,
"可莉": 50,
"流浪者": 51,
"久岐忍": 52,
"神里绫人": 53,
"甘雨": 54,
"戴因斯雷布": 55,
"优菈": 56,
"菲谢尔": 57,
"行秋": 58,
"白术": 59,
"九条裟罗": 60,
"雷泽": 61,
"申鹤": 62,
"迪娜泽黛": 63,
"凯瑟琳": 64,
"多莉": 65,
"坎蒂丝": 66,
"萍姥姥": 67,
"罗莎莉亚": 68,
"留云借风真君": 69,
"绮良良": 70,
"瑶瑶": 71,
"七七": 72,
"奥兹": 73,
"米卡": 74,
"夏洛蒂": 75,
"埃洛伊": 76,
"博士": 77,
"女士": 78,
"大慈树王": 79,
"三月七": 80,
"娜塔莎": 81,
"希露瓦": 82,
"虎克": 83,
"克拉拉": 84,
"丹恒": 85,
"希儿": 86,
"布洛妮娅": 87,
"瓦尔特": 88,
"杰帕德": 89,
"佩拉": 90,
"姬子": 91,
"艾丝妲": 92,
"白露": 93,
"": 94,
"": 95,
"桑博": 96,
"伦纳德": 97,
"停云": 98,
"罗刹": 99,
"卡芙卡": 100,
"彦卿": 101,
"史瓦罗": 102,
"螺丝咕姆": 103,
"阿兰": 104,
"银狼": 105,
"素裳": 106,
"丹枢": 107,
"黑塔": 108,
"景元": 109,
"帕姆": 110,
"可可利亚": 111,
"半夏": 112,
"符玄": 113,
"公输师傅": 114,
"奥列格": 115,
"青雀": 116,
"大毫": 117,
"青镞": 118,
"费斯曼": 119,
"绿芙蓉": 120,
"镜流": 121,
"信使": 122,
"丽塔": 123,
"失落迷迭": 124,
"缭乱星棘": 125,
"伊甸": 126,
"伏特加女孩": 127,
"狂热蓝调": 128,
"莉莉娅": 129,
"萝莎莉娅": 130,
"八重樱": 131,
"八重霞": 132,
"卡莲": 133,
"第六夜想曲": 134,
"卡萝尔": 135,
"极地战刃": 136,
"次生银翼": 137,
"理之律者": 138,
"真理之律者": 139,
"迷城骇兔": 140,
"魇夜星渊": 141,
"黑希儿": 142,
"帕朵菲莉丝": 143,
"天元骑英": 144,
"幽兰黛尔": 145,
"德丽莎": 146,
"月下初拥": 147,
"朔夜观星": 148,
"暮光骑士": 149,
"明日香": 150,
"李素裳": 151,
"格蕾修": 152,
"梅比乌斯": 153,
"渡鸦": 154,
"人之律者": 155,
"爱莉希雅": 156,
"爱衣": 157,
"天穹游侠": 158,
"琪亚娜": 159,
"空之律者": 160,
"终焉之律者": 161,
"薪炎之律者": 162,
"云墨丹心": 163,
"符华": 164,
"识之律者": 165,
"维尔薇": 166,
"始源之律者": 167,
"芽衣": 168,
"雷之律者": 169,
"苏莎娜": 170,
"阿波尼亚": 171,
"陆景和": 172,
"莫弈": 173,
"夏彦": 174,
"左然": 175,
"标贝": 176
}
},
"model": {
@@ -214,14 +247,14 @@
"upsample_rates": [
8,
8,
2,
4,
2
],
"upsample_initial_channel": 512,
"upsample_kernel_sizes": [
16,
16,
4,
8,
4
],
"n_layers_q": 3,

View File

@@ -11,12 +11,15 @@ from utils import load_wav_to_torch, load_filepaths_and_text
from text import cleaned_text_to_sequence, get_bert
"""Multi speaker version"""
class TextAudioSpeakerLoader(torch.utils.data.Dataset):
"""
1) loads audio, speaker_id, text pairs
2) normalizes text and converts them to sequences of integers
3) computes spectrograms from audio files.
"""
def __init__(self, audiopaths_sid_text, hparams):
self.audiopaths_sid_text = load_filepaths_and_text(audiopaths_sid_text)
self.max_wav_value = hparams.max_wav_value
@@ -80,9 +83,9 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
audio_norm = audio / self.max_wav_value
audio_norm = audio_norm.unsqueeze(0)
spec_filename = filename.replace(".wav", ".spec.pt")
if os.path.exists(spec_filename):
try:
spec = torch.load(spec_filename)
else:
except:
spec = spectrogram_torch(audio_norm, self.filter_length,
self.sampling_rate, self.hop_length, self.win_length,
center=False)
@@ -115,8 +118,12 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
assert bert.shape[-1] == len(phone)
except:
bert = get_bert(text, word2ph, language_str)
assert bert.shape[-1] == len(phone), (bert.shape, len(phone), sum(word2ph), p1, p2, t1, t2, pold, pold2,word2ph, text,w2pho)
torch.save(bert, bert_path)
print(bert.shape[-1], bert_path, text, pold)
assert bert.shape[-1] == len(phone)
assert bert.shape[-1] == len(phone), (
bert.shape, len(phone), sum(word2ph), p1, p2, t1, t2, pold, pold2, word2ph, text, w2pho)
phone = torch.LongTensor(phone)
tone = torch.LongTensor(tone)
language = torch.LongTensor(language)
@@ -136,6 +143,7 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
class TextAudioSpeakerCollate():
""" Zero-pads model inputs and targets
"""
def __init__(self, return_ids=False):
self.return_ids = return_ids
@@ -210,6 +218,7 @@ class DistributedBucketSampler(torch.utils.data.distributed.DistributedSampler):
It removes samples which are not included in the boundaries.
Ex) boundaries = [b1, b2, b3] -> any x s.t. length(x) <= b1 or length(x) > b3 are discarded.
"""
def __init__(self, dataset, batch_size, boundaries, num_replicas=None, rank=None, shuffle=True):
super().__init__(dataset, num_replicas=num_replicas, rank=rank, shuffle=shuffle)
self.lengths = dataset.lengths

File diff suppressed because it is too large Load Diff

View File

@@ -1,8 +0,0 @@
SSB00800286|SSB0080|ZH|合肥的-城镇-有什么.|_ h e f ei d e - ch eng zh en - y ou sh en m e . _|0 2 2 2 2 5 5 0 2 2 4 4 0 3 3 2 2 5 5 0 0|1 2 2 2 1 2 2 1 2 2 2 1 1
SSB00800245|SSB0080|ZH|跟-以前的-地方-差不多-一样.|_ g en - y i q ian d e - d i f ang - ch a b u d uo - y i y ang . _|0 1 1 0 3 3 2 2 5 5 0 4 4 5 5 0 4 4 5 5 1 1 0 2 2 4 4 0 0|1 2 1 2 2 2 1 2 2 1 2 2 2 1 2 2 1 1
SSB00120433|SSB0012|ZH|江华的-连续剧-有什么.|_ j iang h ua d e - l ian x v j v - y ou sh en m e . _|0 1 1 2 2 5 5 0 2 2 4 4 4 4 0 3 3 2 2 5 5 0 0|1 2 2 2 1 2 2 2 1 2 2 2 1 1
SSB00120132|SSB0012|ZH|还在你-老爸的-思达-柜台-干活吗.|_ h ai z ai n i - l ao b a d e - s i0 d a - g ui t ai - g an h uo m a . _|0 2 2 4 4 3 3 0 3 3 4 4 5 5 0 1 1 2 2 0 4 4 2 2 0 4 4 2 2 5 5 0 0|1 2 2 2 1 2 2 2 1 2 2 1 2 2 1 2 2 2 1 1
SSB00380151|SSB0038|ZH|给我-放首歌-谭咏麟的歌.|_ g ei w o - f ang sh ou g e - t an y ong l in d e g e . _|0 2 2 3 3 0 4 4 3 3 1 1 0 2 2 3 3 2 2 5 5 1 1 0 0|1 2 2 1 2 2 2 1 2 2 2 2 2 1 1
SSB00380331|SSB0038|ZH|推测-投放-时间为.十九日-下午-或-傍晚.|_ t ui c e - t ou f ang - sh ir j ian w ei . sh ir j iu r ir - x ia w u - h uo - b ang w an . _|0 1 1 4 4 0 2 2 4 4 0 2 2 1 1 4 4 0 2 2 3 3 4 4 0 4 4 3 3 0 4 4 0 4 4 3 3 0 0|1 2 2 1 2 2 1 2 2 2 1 2 2 2 1 2 2 1 2 1 2 2 1 1
SSB03820389|SSB0382|ZH|停车场.|_ t ing ch e ch ang . _|0 2 2 1 1 3 3 0 0|1 2 2 2 1 1
SSB03820048|SSB0382|ZH|宝华-图文.|_ b ao h ua - t u w en . _|0 3 3 2 2 0 2 2 2 2 0 0|1 2 2 1 2 2 1 1

View File

@@ -4,9 +4,9 @@ from random import shuffle
import tqdm
from text.cleaner import clean_text
from collections import defaultdict
stage = [1,2,3]
stage = [2,3]
transcription_path = 'filelists/aishell.list'
transcription_path = 'filelists/genshin.txt'
train_path = 'filelists/train.list'
val_path = 'filelists/val.list'
config_path = "configs/config.json"
@@ -16,11 +16,14 @@ max_val_total = 8
if 1 in stage:
with open( transcription_path+'.cleaned', 'w', encoding='utf-8') as f:
for line in tqdm.tqdm(open(transcription_path, encoding='utf-8').readlines()):
try:
utt, spk, language, text = line.strip().split('|')
norm_text, phones, tones, word2ph = clean_text(text, language)
f.write('{}|{}|{}|{}|{}|{}|{}\n'.format(utt, spk, language, norm_text, ' '.join(phones),
" ".join([str(i) for i in tones]),
" ".join([str(i) for i in word2ph])))
except:
print("err!", text)
if 2 in stage:
spk_utt_map = defaultdict(list)
@@ -34,29 +37,29 @@ if 2 in stage:
if spk not in spk_id_map.keys():
spk_id_map[spk] = current_sid
current_sid += 1
train_list = []
val_list = []
for spk, utts in spk_utt_map.items():
shuffle(utts)
val_list+=utts[:val_per_spk]
train_list+=utts[val_per_spk:]
if len(val_list) > max_val_total:
train_list+=val_list[max_val_total:]
val_list = val_list[:max_val_total]
with open( train_path,"w", encoding='utf-8') as f:
for line in train_list:
f.write(line)
with open(val_path, "w", encoding='utf-8') as f:
for line in val_list:
f.write(line)
#
# train_list = []
# val_list = []
#
# for spk, utts in spk_utt_map.items():
# shuffle(utts)
# val_list+=utts[:val_per_spk]
# train_list+=utts[val_per_spk:]
# if len(val_list) > max_val_total:
# train_list+=val_list[max_val_total:]
# val_list = val_list[:max_val_total]
#
# with open( train_path,"w", encoding='utf-8') as f:
# for line in train_list:
# f.write(line)
#
# with open(val_path, "w", encoding='utf-8') as f:
# for line in val_list:
# f.write(line)
if 3 in stage:
assert 2 in stage
config = json.load(open(config_path))
config["data"]['spk2id'] = spk_id_map
with open(config_path, 'w', encoding='utf-8') as f:
json.dump(config, f, indent=2)
json.dump(config, f, indent=2, ensure_ascii=False)

View File

@@ -1,14 +1,57 @@
import torch
from torch.utils.data import DataLoader
import commons
import utils
from data_utils import TextAudioSpeakerLoader
from data_utils import TextAudioSpeakerLoader, TextAudioSpeakerCollate
from tqdm import tqdm
config_path = 'configs/fzh.json'
from text import cleaned_text_to_sequence, get_bert
config_path = 'configs/config.json'
hps = utils.get_hparams_from_file(config_path)
train_dataset = TextAudioSpeakerLoader(hps.data.training_files, hps.data)
eval_dataset = TextAudioSpeakerLoader(hps.data.validation_files, hps.data)
for _ in tqdm(train_dataset):
collate_fn = TextAudioSpeakerCollate()
train_loader = DataLoader(train_dataset, num_workers=12, shuffle=False,
batch_size=32, pin_memory=True,
drop_last=False, collate_fn=collate_fn)
eval_loader = DataLoader(eval_dataset, num_workers=12, shuffle=False,
batch_size=32, pin_memory=True,
drop_last=False, collate_fn=collate_fn)
for _ in tqdm(train_loader):
pass
for _ in tqdm(eval_dataset):
for _ in tqdm(eval_loader):
pass
# for line in tqdm( open(hps.data.training_files).readlines()):
# _id, spk, language_str, text, phones, tone, word2ph = line.strip().split("|")
# phone = phones.split(" ")
# tone = [int(i) for i in tone.split(" ")]
# word2ph = [int(i) for i in word2ph.split(" ")]
# # print(text, word2ph,phone, tone, language_str)
# w2pho = [i for i in word2ph]
# word2ph = [i for i in word2ph]
# phone, tone, language = cleaned_text_to_sequence(phone, tone, language_str)
# pold2 = phone
# if hps.data.add_blank:
# phone = commons.intersperse(phone, 0)
# tone = commons.intersperse(tone, 0)
# language = commons.intersperse(language, 0)
# for i in range(len(word2ph)):
# word2ph[i] = word2ph[i] * 2
# word2ph[0] += 1
# wav_path = f'dataset/{spk}/{_id}.wav'
# bert_path = wav_path.replace(".wav", ".bert.pt")
# try:
# bert = torch.load(bert_path)
# assert bert.shape[-1] == len(phone)
# except:
# bert = get_bert(text, word2ph, language_str)
# assert bert.shape[-1] == len(phone)
# torch.save(bert, bert_path)

View File

@@ -72,7 +72,7 @@ def run(rank, n_gpus, hps):
rank=rank,
shuffle=True)
collate_fn = TextAudioSpeakerCollate()
train_loader = DataLoader(train_dataset, num_workers=4, shuffle=False, pin_memory=True,
train_loader = DataLoader(train_dataset, num_workers=20, shuffle=False, pin_memory=True,
collate_fn=collate_fn, batch_sampler=train_sampler, persistent_workers=True)
if rank == 0:
eval_dataset = TextAudioSpeakerLoader(hps.data.validation_files, hps.data)
@@ -107,7 +107,7 @@ def run(rank, n_gpus, hps):
net_g = DDP(net_g, device_ids=[rank])
net_d = DDP(net_d, device_ids=[rank])
pretrain_dir = "logs/esd"
pretrain_dir = None
if pretrain_dir is None:
_, _, _, epoch_str = utils.load_checkpoint(utils.latest_checkpoint_path(hps.model_dir, "G_*.pth"), net_g,
optim_g, False)
@@ -120,8 +120,6 @@ def run(rank, n_gpus, hps):
optim_g, True)
_, _, _, epoch_str = utils.load_checkpoint(utils.latest_checkpoint_path(pretrain_dir, "D_*.pth"), net_d,
optim_d, True)
epoch_str = 1
global_step = 0