Dev 2.3. (#242)
* Fix inputs of duration discriminator * Add LSTM * Update models.py * Update tensorboard scalar * Noise injection for minimizing modality gap * Update infer.py * support bf16 run * del unused_para flag * support bf16 config * add grad clip * fix(logger and grad):add dur grad,fix grad clip * Update webui_preprocess.py * Fix English G2P * fix(bert_gen):add pass * Pass SDP to DD * Update webui_preprocess.py * Update config.json * Update webui.py * Update chinese_bert.py * Upload webui for deploy * Update webui.py * torch.save as pt not npy * Update config.json * add freeze emo vq * Update webui_preprocess.py * Fix tone_sandhi.py * Comment up grad clip * Fix in-place addition * Add SLM discriminator * Add DDP for WD * Feat: Style text: make emotions and style similar to the style text by mixing bert (#240) (#241) * fix:(oldVersion210) Load on demand Emotion model * feat: update fastapi.py. 添加更多错误日志信息 * Switch pyopenjtalk to pyopenjtalk-prebuilt * fix: update fastapi.py. 2.2 reference适配 * Update resample.py * 修复Onnx导出的BUG (#237) * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Add files via upload * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Delete attentions_onnx.py * Delete models_onnx.py * Add files via upload * Add files via upload * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Update __init__.py * Update __init__.py * Update __init__.py * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- * Fix onnx * Format export * Feat: style-text and bert mixing (JA only) * Ensure the same tensor shape * Update * update gradio version * Fix * Style text for chinese and english (ver 2.2) * Style text for chinese and english (ver 2.1) * Style text in FastAPI * Translate style text desc in chinese --------- Co-authored-by: litagin02 <139731664+litagin02@users.noreply.github.com> Co-authored-by: Sora <654163754@qq.com> Co-authored-by: Sihan Wang <wangsihan1995@gmail.com> Co-authored-by: Ναρουσέ·μ·γιουμεμί·Χινακάννα <40709280+NaruseMioShirakana@users.noreply.github.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> * Remove CLAP * Revert "Remove CLAP" This reverts commit 62fd59bc837c580239840a2bc84b15e0663730fc. Revert * Remove CLAP * bf16 audo grad cilp * Update webui and infer utils * Update webui.py * Update webui.py * Update webui-preprocess.py * Update webui_preprocess.py --------- Co-authored-by: Sihan Wang <wangsihan1995@gmail.com> Co-authored-by: OedoSoldier <31711261+OedoSoldier@users.noreply.github.com> Co-authored-by: litagin02 <139731664+litagin02@users.noreply.github.com> Co-authored-by: Sora <654163754@qq.com> Co-authored-by: Ναρουσέ·μ·γιουμεμί·Χινακάννα <40709280+NaruseMioShirakana@users.noreply.github.com> Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
This commit is contained in:
@@ -44,10 +44,6 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
|
||||
self.min_text_len = getattr(hparams, "min_text_len", 1)
|
||||
self.max_text_len = getattr(hparams, "max_text_len", 384)
|
||||
|
||||
self.empty_emo = torch.squeeze(
|
||||
torch.load("empty_emo.npy", map_location="cpu"), dim=1
|
||||
)
|
||||
|
||||
random.seed(1234)
|
||||
random.shuffle(self.audiopaths_sid_text)
|
||||
self._filter()
|
||||
@@ -98,14 +94,7 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
|
||||
spec, wav = self.get_audio(audiopath)
|
||||
sid = torch.LongTensor([int(self.spk_map[sid])])
|
||||
|
||||
if np.random.rand() > 0.1:
|
||||
emo = torch.squeeze(
|
||||
torch.load(audiopath.replace(".wav", ".emo.npy"), map_location="cpu"),
|
||||
dim=1,
|
||||
)
|
||||
else:
|
||||
emo = self.empty_emo
|
||||
return (phones, spec, wav, sid, tone, language, bert, ja_bert, en_bert, emo)
|
||||
return (phones, spec, wav, sid, tone, language, bert, ja_bert, en_bert)
|
||||
|
||||
def get_audio(self, filename):
|
||||
audio, sampling_rate = load_wav_to_torch(filename)
|
||||
@@ -168,15 +157,15 @@ class TextAudioSpeakerLoader(torch.utils.data.Dataset):
|
||||
|
||||
if language_str == "ZH":
|
||||
bert = bert_ori
|
||||
ja_bert = torch.rand(1024, len(phone))
|
||||
en_bert = torch.rand(1024, len(phone))
|
||||
ja_bert = torch.randn(1024, len(phone))
|
||||
en_bert = torch.randn(1024, len(phone))
|
||||
elif language_str == "JP":
|
||||
bert = torch.rand(1024, len(phone))
|
||||
bert = torch.randn(1024, len(phone))
|
||||
ja_bert = bert_ori
|
||||
en_bert = torch.rand(1024, len(phone))
|
||||
en_bert = torch.randn(1024, len(phone))
|
||||
elif language_str == "EN":
|
||||
bert = torch.rand(1024, len(phone))
|
||||
ja_bert = torch.rand(1024, len(phone))
|
||||
bert = torch.randn(1024, len(phone))
|
||||
ja_bert = torch.randn(1024, len(phone))
|
||||
en_bert = bert_ori
|
||||
phone = torch.LongTensor(phone)
|
||||
tone = torch.LongTensor(tone)
|
||||
@@ -226,7 +215,6 @@ class TextAudioSpeakerCollate:
|
||||
bert_padded = torch.FloatTensor(len(batch), 1024, max_text_len)
|
||||
ja_bert_padded = torch.FloatTensor(len(batch), 1024, max_text_len)
|
||||
en_bert_padded = torch.FloatTensor(len(batch), 1024, max_text_len)
|
||||
emo = torch.FloatTensor(len(batch), 512)
|
||||
|
||||
spec_padded = torch.FloatTensor(len(batch), batch[0][1].size(0), max_spec_len)
|
||||
wav_padded = torch.FloatTensor(len(batch), 1, max_wav_len)
|
||||
@@ -238,7 +226,6 @@ class TextAudioSpeakerCollate:
|
||||
bert_padded.zero_()
|
||||
ja_bert_padded.zero_()
|
||||
en_bert_padded.zero_()
|
||||
emo.zero_()
|
||||
|
||||
for i in range(len(ids_sorted_decreasing)):
|
||||
row = batch[ids_sorted_decreasing[i]]
|
||||
@@ -272,8 +259,6 @@ class TextAudioSpeakerCollate:
|
||||
en_bert = row[8]
|
||||
en_bert_padded[i, :, : en_bert.size(1)] = en_bert
|
||||
|
||||
emo[i, :] = row[9]
|
||||
|
||||
return (
|
||||
text_padded,
|
||||
text_lengths,
|
||||
@@ -287,7 +272,6 @@ class TextAudioSpeakerCollate:
|
||||
bert_padded,
|
||||
ja_bert_padded,
|
||||
en_bert_padded,
|
||||
emo,
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user