diff --git a/tts/StyleTTS2Train/Models/LJSpeech/config_harness.yml b/tts/StyleTTS2Train/Models/LJSpeech/config_harness.yml new file mode 100644 index 0000000..42900db --- /dev/null +++ b/tts/StyleTTS2Train/Models/LJSpeech/config_harness.yml @@ -0,0 +1,110 @@ +log_dir: "Models/LJSpeech" +save_freq: 5 +log_interval: 10 +device: "cuda" +epochs: 50 # number of finetuning epoch (1 hour of data) +batch_size: 8 +max_len: 400 # maximum number of frames +pretrained_model: "Models/LibriTTS/epochs_2nd_00020.pth" +second_stage_load_pretrained: true # set to true if the pre-trained model is for 2nd stage +load_only_params: true # set to true if do not want to load epoch numbers and optimizer parameters + +F0_path: "Utils/JDC/bst.t7" +ASR_config: "Utils/ASR/config.yml" +ASR_path: "Utils/ASR/epoch_00080.pth" +PLBERT_dir: 'Utils/PLBERT/' + +data_params: + train_data: "Data/train_harness.txt" + val_data: "Data/val_harness.txt" + root_path: "" + OOD_data: "Data/OOD_harness.txt" + min_length: 50 # sample until texts with this size are obtained for OOD texts + +preprocess_params: + sr: 24000 + spect_params: + n_fft: 2048 + win_length: 1200 + hop_length: 300 + +model_params: + multispeaker: true + + dim_in: 64 + hidden_dim: 512 + max_conv_dim: 512 + n_layer: 3 + n_mels: 80 + + n_token: 178 # number of phoneme tokens + max_dur: 50 # maximum duration of a single phoneme + style_dim: 128 # style vector size + + dropout: 0.2 + + # config for decoder + decoder: + type: 'hifigan' # either hifigan or istftnet + resblock_kernel_sizes: [3,7,11] + upsample_rates : [10,5,3,2] + upsample_initial_channel: 512 + resblock_dilation_sizes: [[1,3,5], [1,3,5], [1,3,5]] + upsample_kernel_sizes: [20,10,6,4] + + # speech language model config + slm: + model: 'microsoft/wavlm-base-plus' + sr: 16000 # sampling rate of SLM + hidden: 768 # hidden size of SLM + nlayers: 13 # number of layers of SLM + initial_channel: 64 # initial channels of SLM discriminator head + + # style diffusion model config + diffusion: + embedding_mask_proba: 0.1 + # transformer config + transformer: + num_layers: 3 + num_heads: 8 + head_features: 64 + multiplier: 2 + + # diffusion distribution config + dist: + sigma_data: 0.2 # placeholder for estimate_sigma_data set to false + estimate_sigma_data: true # estimate sigma_data from the current batch if set to true + mean: -3.0 + std: 1.0 + +loss_params: + lambda_mel: 5. # mel reconstruction loss + lambda_gen: 1. # generator loss + lambda_slm: 1. # slm feature matching loss + + lambda_mono: 1. # monotonic alignment loss (TMA) + lambda_s2s: 1. # sequence-to-sequence loss (TMA) + + lambda_F0: 1. # F0 reconstruction loss + lambda_norm: 1. # norm reconstruction loss + lambda_dur: 1. # duration loss + lambda_ce: 20. # duration predictor probability output CE loss + lambda_sty: 1. # style reconstruction loss + lambda_diff: 1. # score matching loss + + diff_epoch: 10 # style diffusion starting epoch + joint_epoch: 30 # joint training starting epoch + +optimizer_params: + lr: 0.0001 # general learning rate + bert_lr: 0.00001 # learning rate for PLBERT + ft_lr: 0.0001 # learning rate for acoustic modules + +slmadv_params: + min_len: 400 # minimum length of samples + max_len: 500 # maximum length of samples + batch_percentage: 0.5 # to prevent out of memory, only use half of the original batch size + iter: 10 # update the discriminator every this iterations of generator update + thresh: 5 # gradient norm above which the gradient is scaled + scale: 0.01 # gradient scaling factor for predictors from SLM discriminators + sig: 1.5 # sigma for differentiable duration modeling \ No newline at end of file diff --git a/tts/StyleTTS2Train/Models/LJSpeech/train.log b/tts/StyleTTS2Train/Models/LJSpeech/train.log new file mode 100644 index 0000000..e69de29 diff --git a/tts/StyleTTS2Train/Models/LibriTTS/config_harness.yml b/tts/StyleTTS2Train/Models/LibriTTS/config_harness.yml new file mode 100644 index 0000000..7525632 --- /dev/null +++ b/tts/StyleTTS2Train/Models/LibriTTS/config_harness.yml @@ -0,0 +1,111 @@ +log_dir: "Models/LJSpeech" +save_freq: 5 +log_interval: 10 +device: "cuda" +epochs: 50 # number of finetuning epoch (1 hour of data) +batch_size: 1 +max_len: 400 # maximum number of frames +pretrained_model: "Models/LibriTTS/epochs_2nd_00020.pth" +second_stage_load_pretrained: true # set to true if the pre-trained model is for 2nd stage +load_only_params: true # set to true if do not want to load epoch numbers and optimizer parameters + +F0_path: "Utils/JDC/bst.t7" +ASR_config: "Utils/ASR/config.yml" +ASR_path: "Utils/ASR/epoch_00080.pth" +PLBERT_dir: 'Utils/PLBERT/' + +data_params: + train_data: "Data/train_harness.txt" + val_data: "Data/val_harness.txt" + root_path: "" + OOD_data: "Data/OOD_harness.txt" + min_length: 50 # sample until texts with this size are obtained for OOD texts + +preprocess_params: + sr: 24000 + spect_params: + n_fft: 2048 + win_length: 1200 + hop_length: 300 + +model_params: + multispeaker: true + + dim_in: 64 + hidden_dim: 512 + max_conv_dim: 512 + n_layer: 3 + n_mels: 80 + + n_token: 178 # number of phoneme tokens + max_dur: 50 # maximum duration of a single phoneme + style_dim: 128 # style vector size + + dropout: 0.2 + + # config for decoder + decoder: + type: 'hifigan' # either hifigan or istftnet + resblock_kernel_sizes: [3,7,11] + upsample_rates : [10,5,3,2] + upsample_initial_channel: 512 + resblock_dilation_sizes: [[1,3,5], [1,3,5], [1,3,5]] + upsample_kernel_sizes: [20,10,6,4] + + # speech language model config + slm: + model: 'microsoft/wavlm-base-plus' + sr: 16000 # sampling rate of SLM + hidden: 768 # hidden size of SLM + nlayers: 13 # number of layers of SLM + initial_channel: 64 # initial channels of SLM discriminator head + + # style diffusion model config + diffusion: + embedding_mask_proba: 0.1 + # transformer config + transformer: + num_layers: 3 + num_heads: 8 + head_features: 64 + multiplier: 2 + + # diffusion distribution config + dist: + sigma_data: 0.2 # placeholder for estimate_sigma_data set to false + estimate_sigma_data: true # estimate sigma_data from the current batch if set to true + mean: -3.0 + std: 1.0 + +loss_params: + lambda_mel: 5. # mel reconstruction loss + lambda_gen: 1. # generator loss + lambda_slm: 1. # slm feature matching loss + + lambda_mono: 1. # monotonic alignment loss (TMA) + lambda_s2s: 1. # sequence-to-sequence loss (TMA) + + lambda_F0: 1. # F0 reconstruction loss + lambda_norm: 1. # norm reconstruction loss + lambda_dur: 1. # duration loss + lambda_ce: 20. # duration predictor probability output CE loss + lambda_sty: 1. # style reconstruction loss + lambda_diff: 1. # score matching loss + + diff_epoch: 10 # style diffusion starting epoch + joint_epoch: 30 # joint training starting epoch + +optimizer_params: + lr: 0.0001 # general learning rate + bert_lr: 0.00001 # learning rate for PLBERT + ft_lr: 0.0001 # learning rate for acoustic modules + +slmadv_params: + min_len: 400 # minimum length of samples + max_len: 500 # maximum length of samples + batch_percentage: 0.5 # to prevent out of memory, only use half of the original batch size + iter: 10 # update the discriminator every this iterations of generator update + thresh: 5 # gradient norm above which the gradient is scaled + scale: 0.01 # gradient scaling factor for predictors from SLM discriminators + sig: 1.5 # sigma for differentiable duration modeling + \ No newline at end of file diff --git a/tts/StyleTTS2Train/Models/LibriTTS/train.log b/tts/StyleTTS2Train/Models/LibriTTS/train.log new file mode 100644 index 0000000..e69de29 diff --git a/tts/StyleTTS2Train/TestHarnessV2/TestHarnessV2.zip b/tts/StyleTTS2Train/TestHarnessV2/TestHarnessV2.zip new file mode 100644 index 0000000..5dc1527 Binary files /dev/null and b/tts/StyleTTS2Train/TestHarnessV2/TestHarnessV2.zip differ diff --git a/tts/StyleTTS2TrainOG/train_finetune.py b/tts/StyleTTS2TrainOG/train_finetune.py index 3c65074..1374719 100644 --- a/tts/StyleTTS2TrainOG/train_finetune.py +++ b/tts/StyleTTS2TrainOG/train_finetune.py @@ -46,6 +46,8 @@ handler = StreamHandler() handler.setLevel(logging.DEBUG) logger.addHandler(handler) +from test_harness import TestHarness +harness = TestHarness() @click.command() @click.option('-p', '--config_path', default='Configs/config_ft.yml', type=str) @@ -559,6 +561,13 @@ def main(config_path): writer.add_scalar('train/gen_loss_slm', loss_gen_lm, iters) running_loss = 0 + step_total = len(train_list)//batch_size + train_list_len = len(train_list) + step = i+1 + + harness.log_values(loss_gen_all, d_loss, loss_ce, loss_dur, loss_lm, loss_norm_rec, + loss_F0_rec, loss_sty, loss_diff, d_loss_slm, loss_gen_lm, iters, epoch, step, + log_interval, running_loss,train_list_len,batch_size) print('Time elasped:', time.time()-start_time) @@ -680,6 +689,7 @@ def main(config_path): writer.add_scalar('eval/dur_loss', loss_test / iters_test, epoch + 1) writer.add_scalar('eval/F0_loss', loss_f / iters_test, epoch + 1) + harness.log_eval(mel_loss=float(loss_test / iters_test),dur_loss=float(loss_test / iters_test),F0_loss=float(loss_f / iters_test),epoch=epoch) if (epoch + 1) % save_freq == 0 : if (loss_test / iters_test) < best_loss: @@ -694,7 +704,8 @@ def main(config_path): } save_path = osp.join(log_dir, 'epoch_2nd_%05d.pth' % epoch) torch.save(state, save_path) - + + harness.test(save_path) # if estimate sigma, save the estimated simga if model_params.diffusion.dist.estimate_sigma_data: config['model_params']['diffusion']['dist']['sigma_data'] = float(np.mean(running_std))