mirror of
https://github.com/storytold/vits-finetuning.git
synced 2026-10-09 00:09:52 +00:00
Make easier to train on max sample length = 10s
This commit is contained in:
@@ -8,8 +8,8 @@
|
|||||||
"learning_rate": 2e-4,
|
"learning_rate": 2e-4,
|
||||||
"betas": [0.8, 0.99],
|
"betas": [0.8, 0.99],
|
||||||
"eps": 1e-9,
|
"eps": 1e-9,
|
||||||
"batch_size": 16,
|
"batch_size": 12,
|
||||||
"grad_acc_steps": 2,
|
"grad_acc_steps": 3,
|
||||||
"fp16_run": true,
|
"fp16_run": true,
|
||||||
"lr_decay": 0.999875,
|
"lr_decay": 0.999875,
|
||||||
"segment_size": 8192,
|
"segment_size": 8192,
|
||||||
|
|||||||
+1
-1
@@ -58,7 +58,7 @@ class TextAudioLoader(torch.utils.data.Dataset):
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
au_len = librosa.get_duration(filename=audiopath)
|
au_len = librosa.get_duration(filename=audiopath)
|
||||||
if au_len > 13.0:
|
if au_len > 10.0:
|
||||||
n_fau += 1
|
n_fau += 1
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -24,7 +24,7 @@ if __name__ == '__main__':
|
|||||||
test_sentences = ["The quick brown fox jumps over the lazy dog",
|
test_sentences = ["The quick brown fox jumps over the lazy dog",
|
||||||
"In a galaxy far, far away, a young hero embarks on an epic adventure",
|
"In a galaxy far, far away, a young hero embarks on an epic adventure",
|
||||||
"Ladies and gentlemen, welcome to the annual science fair!",
|
"Ladies and gentlemen, welcome to the annual science fair!",
|
||||||
"The chef skillfully prepares a delicious gourmet meal with fresh ingredients and exquisite flavors",
|
"Peter Piper picked a peck of pickled peppers. How many pickled peppers did Peter Piper pick?",
|
||||||
"The crowd erupted in cheers as the team scored the winning goal in the final seconds of the match"]
|
"The crowd erupted in cheers as the team scored the winning goal in the final seconds of the match"]
|
||||||
|
|
||||||
if len(args.test) > 1:
|
if len(args.test) > 1:
|
||||||
|
|||||||
@@ -394,7 +394,9 @@ def evaluate(hps, generator, eval_loader, writer_eval):
|
|||||||
t_bert_lens = torch.LongTensor([t_bert.size(1)]).cuda(0)
|
t_bert_lens = torch.LongTensor([t_bert.size(1)]).cuda(0)
|
||||||
t_bert = t_bert.cuda(0)
|
t_bert = t_bert.cuda(0)
|
||||||
|
|
||||||
audio, attn, t_mask, _ = generator.module.infer(t_text_norm, t_text_lengths, t_moji, t_bert, t_bert_lens, noise_scale=.667, noise_scale_w=0.8, length_scale=1.0)
|
with torch.no_grad():
|
||||||
|
audio, attn, t_mask, _ = generator.module.infer(t_text_norm, t_text_lengths, t_moji, t_bert, t_bert_lens, noise_scale=.667, noise_scale_w=0.8, length_scale=1.0)
|
||||||
|
|
||||||
test_audio_lengths = t_mask.sum([1,2]).long() * hps.data.hop_length
|
test_audio_lengths = t_mask.sum([1,2]).long() * hps.data.hop_length
|
||||||
y_test_mel = mel_spectrogram_torch(
|
y_test_mel = mel_spectrogram_torch(
|
||||||
audio.squeeze(1).float(),
|
audio.squeeze(1).float(),
|
||||||
|
|||||||
Reference in New Issue
Block a user