mirror of
https://github.com/storytold/storyteller-ml.git
synced 2026-10-09 00:09:55 +00:00
Merge pull request #20 from storytold/brandon-rmvpe-support
Rmvpe support
This commit is contained in:
@@ -119,6 +119,8 @@ parser.add_argument('--model_path', type=str, help='path to the .pth file', requ
|
||||
parser.add_argument('--model_index_path', type=str, help='path to the .index file (empty string for none)', required=False, default='')
|
||||
parser.add_argument('--hubert_model_path', type=str, help='path to the hubert model .pth file (default hubert_base.pt)',
|
||||
required=False, default='hubert_base.pt')
|
||||
parser.add_argument('--rmvpe_model_path', type=str, help='path to the RMVPE model .pth file (default rmvpe.pt)',
|
||||
required=False, default='rmvpe.pt')
|
||||
|
||||
parser.add_argument('--input_audio_filename', type=str, help='source audio file', required=True)
|
||||
parser.add_argument('--output_audio_filename', type=str, help='result audio file', required=True)
|
||||
@@ -130,7 +132,7 @@ parser.add_argument('--f0_up_key', type=str,
|
||||
help='the f0 key for the input audio file(default "0")',
|
||||
required=False, default='0')
|
||||
parser.add_argument('--f0_method', type=str,
|
||||
help='f0 estimation method to use: harvest (default), crepe, or pm',
|
||||
help='f0 estimation method to use: rmvpe (best), harvest (default), crepe, or pm',
|
||||
required=False, default='harvest')
|
||||
parser.add_argument('--index_rate', type=float,
|
||||
help='The rate for the index (search feature ratio) (default 0.75)',
|
||||
@@ -168,6 +170,7 @@ resample_sr = args.resample_sr
|
||||
rms_mix_rate = args.rms_mix_rate
|
||||
protect = args.protect
|
||||
hubert_path = args.hubert_model_path
|
||||
rmvpe_path = args.rmvpe_model_path
|
||||
|
||||
config = Config(device, is_half)
|
||||
now_dir = os.getcwd()
|
||||
@@ -201,7 +204,7 @@ def load_hubert(hubert_path="hubert_base.pt"):
|
||||
hubert_model.eval()
|
||||
|
||||
|
||||
def vc_single(sid, input_audio, f0_up_key, f0_file, f0_method, file_index, index_rate, hubert_path="hubert_base.pt"):
|
||||
def vc_single(sid, input_audio, f0_up_key, f0_file, f0_method, file_index, index_rate, rmvpe_path, hubert_path="hubert_base.pt"):
|
||||
global tgt_sr, net_g, vc, hubert_model, version
|
||||
if input_audio is None:
|
||||
return "You need to upload an audio", None
|
||||
@@ -218,6 +221,7 @@ def vc_single(sid, input_audio, f0_up_key, f0_file, f0_method, file_index, index
|
||||
sid,
|
||||
audio,
|
||||
input_audio,
|
||||
rmvpe_path,
|
||||
times,
|
||||
f0_up_key,
|
||||
f0_method,
|
||||
@@ -270,7 +274,7 @@ get_vc(model_path)
|
||||
|
||||
file_path = input_path
|
||||
wav_opt = vc_single(
|
||||
0, file_path, f0up_key, None, f0method, index_path, index_rate, hubert_path
|
||||
0, file_path, f0up_key, None, f0method, index_path, index_rate, rmvpe_path, hubert_path
|
||||
)
|
||||
out_path = opt_path
|
||||
wavfile.write(out_path, tgt_sr, wav_opt)
|
||||
|
||||
@@ -1,139 +1,139 @@
|
||||
import sys, os, multiprocessing
|
||||
from scipy import signal
|
||||
|
||||
now_dir = os.getcwd()
|
||||
sys.path.append(now_dir)
|
||||
print(sys.argv)
|
||||
inp_root = sys.argv[1]
|
||||
sr = int(sys.argv[2])
|
||||
n_p = int(sys.argv[3])
|
||||
exp_dir = sys.argv[4]
|
||||
noparallel = sys.argv[5] == "True"
|
||||
import numpy as np, os, traceback
|
||||
from lib.slicer2 import Slicer
|
||||
import librosa, traceback
|
||||
from scipy.io import wavfile
|
||||
import multiprocessing
|
||||
from lib.audio import load_audio
|
||||
|
||||
mutex = multiprocessing.Lock()
|
||||
f = open("%s/preprocess.log" % exp_dir, "a+")
|
||||
|
||||
|
||||
def println(strr):
|
||||
mutex.acquire()
|
||||
print(strr)
|
||||
f.write("%s\n" % strr)
|
||||
f.flush()
|
||||
mutex.release()
|
||||
|
||||
|
||||
class PreProcess:
|
||||
def __init__(self, sr, exp_dir):
|
||||
self.slicer = Slicer(
|
||||
sr=sr,
|
||||
threshold=-42,
|
||||
min_length=1500,
|
||||
min_interval=400,
|
||||
hop_size=15,
|
||||
max_sil_kept=500,
|
||||
)
|
||||
self.sr = sr
|
||||
self.bh, self.ah = signal.butter(N=5, Wn=48, btype="high", fs=self.sr)
|
||||
self.per = 3.0
|
||||
self.overlap = 0.3
|
||||
self.tail = self.per + self.overlap
|
||||
self.max = 0.9
|
||||
self.alpha = 0.75
|
||||
self.exp_dir = exp_dir
|
||||
self.gt_wavs_dir = "%s/0_gt_wavs" % exp_dir
|
||||
self.wavs16k_dir = "%s/1_16k_wavs" % exp_dir
|
||||
os.makedirs(self.exp_dir, exist_ok=True)
|
||||
os.makedirs(self.gt_wavs_dir, exist_ok=True)
|
||||
os.makedirs(self.wavs16k_dir, exist_ok=True)
|
||||
|
||||
def norm_write(self, tmp_audio, idx0, idx1):
|
||||
tmp_max = np.abs(tmp_audio).max()
|
||||
if tmp_max > 2.5:
|
||||
print("%s-%s-%s-filtered" % (idx0, idx1, tmp_max))
|
||||
return
|
||||
tmp_audio = (tmp_audio / tmp_max * (self.max * self.alpha)) + (
|
||||
1 - self.alpha
|
||||
) * tmp_audio
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.gt_wavs_dir, idx0, idx1),
|
||||
self.sr,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
tmp_audio = librosa.resample(
|
||||
tmp_audio, orig_sr=self.sr, target_sr=16000
|
||||
) # , res_type="soxr_vhq"
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.wavs16k_dir, idx0, idx1),
|
||||
16000,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
|
||||
def pipeline(self, path, idx0):
|
||||
try:
|
||||
audio = load_audio(path, self.sr)
|
||||
# zero phased digital filter cause pre-ringing noise...
|
||||
# audio = signal.filtfilt(self.bh, self.ah, audio)
|
||||
audio = signal.lfilter(self.bh, self.ah, audio)
|
||||
|
||||
idx1 = 0
|
||||
for audio in self.slicer.slice(audio):
|
||||
i = 0
|
||||
while 1:
|
||||
start = int(self.sr * (self.per - self.overlap) * i)
|
||||
i += 1
|
||||
if len(audio[start:]) > self.tail * self.sr:
|
||||
tmp_audio = audio[start : start + int(self.per * self.sr)]
|
||||
self.norm_write(tmp_audio, idx0, idx1)
|
||||
idx1 += 1
|
||||
else:
|
||||
tmp_audio = audio[start:]
|
||||
idx1 += 1
|
||||
break
|
||||
self.norm_write(tmp_audio, idx0, idx1)
|
||||
println("%s->Suc." % path)
|
||||
except:
|
||||
println("%s->%s" % (path, traceback.format_exc()))
|
||||
|
||||
def pipeline_mp(self, infos):
|
||||
for path, idx0 in infos:
|
||||
self.pipeline(path, idx0)
|
||||
|
||||
def pipeline_mp_inp_dir(self, inp_root, n_p):
|
||||
try:
|
||||
infos = [
|
||||
("%s/%s" % (inp_root, name), idx)
|
||||
for idx, name in enumerate(sorted(list(os.listdir(inp_root))))
|
||||
]
|
||||
if noparallel:
|
||||
for i in range(n_p):
|
||||
self.pipeline_mp(infos[i::n_p])
|
||||
else:
|
||||
ps = []
|
||||
for i in range(n_p):
|
||||
p = multiprocessing.Process(
|
||||
target=self.pipeline_mp, args=(infos[i::n_p],)
|
||||
)
|
||||
ps.append(p)
|
||||
p.start()
|
||||
for i in range(n_p):
|
||||
ps[i].join()
|
||||
except:
|
||||
println("Fail. %s" % traceback.format_exc())
|
||||
|
||||
|
||||
def preprocess_trainset(inp_root, sr, n_p, exp_dir):
|
||||
pp = PreProcess(sr, exp_dir)
|
||||
println("start preprocess")
|
||||
println(sys.argv)
|
||||
pp.pipeline_mp_inp_dir(inp_root, n_p)
|
||||
println("end preprocess")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
preprocess_trainset(inp_root, sr, n_p, exp_dir)
|
||||
import sys, os, multiprocessing
|
||||
from scipy import signal
|
||||
|
||||
now_dir = os.getcwd()
|
||||
sys.path.append(now_dir)
|
||||
print(sys.argv)
|
||||
inp_root = sys.argv[1]
|
||||
sr = int(sys.argv[2])
|
||||
n_p = int(sys.argv[3])
|
||||
exp_dir = sys.argv[4]
|
||||
noparallel = sys.argv[5] == "True"
|
||||
import numpy as np, os, traceback
|
||||
from lib.slicer2 import Slicer
|
||||
import librosa, traceback
|
||||
from scipy.io import wavfile
|
||||
import multiprocessing
|
||||
from lib.audio import load_audio
|
||||
|
||||
mutex = multiprocessing.Lock()
|
||||
f = open("%s/preprocess.log" % exp_dir, "a+")
|
||||
|
||||
|
||||
def println(strr):
|
||||
mutex.acquire()
|
||||
print(strr)
|
||||
f.write("%s\n" % strr)
|
||||
f.flush()
|
||||
mutex.release()
|
||||
|
||||
|
||||
class PreProcess:
|
||||
def __init__(self, sr, exp_dir):
|
||||
self.slicer = Slicer(
|
||||
sr=sr,
|
||||
threshold=-42,
|
||||
min_length=1500,
|
||||
min_interval=400,
|
||||
hop_size=15,
|
||||
max_sil_kept=500,
|
||||
)
|
||||
self.sr = sr
|
||||
self.bh, self.ah = signal.butter(N=5, Wn=48, btype="high", fs=self.sr)
|
||||
self.per = 3.0
|
||||
self.overlap = 0.3
|
||||
self.tail = self.per + self.overlap
|
||||
self.max = 0.9
|
||||
self.alpha = 0.75
|
||||
self.exp_dir = exp_dir
|
||||
self.gt_wavs_dir = "%s/0_gt_wavs" % exp_dir
|
||||
self.wavs16k_dir = "%s/1_16k_wavs" % exp_dir
|
||||
os.makedirs(self.exp_dir, exist_ok=True)
|
||||
os.makedirs(self.gt_wavs_dir, exist_ok=True)
|
||||
os.makedirs(self.wavs16k_dir, exist_ok=True)
|
||||
|
||||
def norm_write(self, tmp_audio, idx0, idx1):
|
||||
tmp_max = np.abs(tmp_audio).max()
|
||||
if tmp_max > 2.5:
|
||||
print("%s-%s-%s-filtered" % (idx0, idx1, tmp_max))
|
||||
return
|
||||
tmp_audio = (tmp_audio / tmp_max * (self.max * self.alpha)) + (
|
||||
1 - self.alpha
|
||||
) * tmp_audio
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.gt_wavs_dir, idx0, idx1),
|
||||
self.sr,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
tmp_audio = librosa.resample(
|
||||
tmp_audio, orig_sr=self.sr, target_sr=16000
|
||||
) # , res_type="soxr_vhq"
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.wavs16k_dir, idx0, idx1),
|
||||
16000,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
|
||||
def pipeline(self, path, idx0):
|
||||
try:
|
||||
audio = load_audio(path, self.sr)
|
||||
# zero phased digital filter cause pre-ringing noise...
|
||||
# audio = signal.filtfilt(self.bh, self.ah, audio)
|
||||
audio = signal.lfilter(self.bh, self.ah, audio)
|
||||
|
||||
idx1 = 0
|
||||
for audio in self.slicer.slice(audio):
|
||||
i = 0
|
||||
while 1:
|
||||
start = int(self.sr * (self.per - self.overlap) * i)
|
||||
i += 1
|
||||
if len(audio[start:]) > self.tail * self.sr:
|
||||
tmp_audio = audio[start : start + int(self.per * self.sr)]
|
||||
self.norm_write(tmp_audio, idx0, idx1)
|
||||
idx1 += 1
|
||||
else:
|
||||
tmp_audio = audio[start:]
|
||||
idx1 += 1
|
||||
break
|
||||
self.norm_write(tmp_audio, idx0, idx1)
|
||||
println("%s->Suc." % path)
|
||||
except:
|
||||
println("%s->%s" % (path, traceback.format_exc()))
|
||||
|
||||
def pipeline_mp(self, infos):
|
||||
for path, idx0 in infos:
|
||||
self.pipeline(path, idx0)
|
||||
|
||||
def pipeline_mp_inp_dir(self, inp_root, n_p):
|
||||
try:
|
||||
infos = [
|
||||
("%s/%s" % (inp_root, name), idx)
|
||||
for idx, name in enumerate(sorted(list(os.listdir(inp_root))))
|
||||
]
|
||||
if noparallel:
|
||||
for i in range(n_p):
|
||||
self.pipeline_mp(infos[i::n_p])
|
||||
else:
|
||||
ps = []
|
||||
for i in range(n_p):
|
||||
p = multiprocessing.Process(
|
||||
target=self.pipeline_mp, args=(infos[i::n_p],)
|
||||
)
|
||||
ps.append(p)
|
||||
p.start()
|
||||
for i in range(n_p):
|
||||
ps[i].join()
|
||||
except:
|
||||
println("Fail. %s" % traceback.format_exc())
|
||||
|
||||
|
||||
def preprocess_trainset(inp_root, sr, n_p, exp_dir):
|
||||
pp = PreProcess(sr, exp_dir)
|
||||
println("start preprocess")
|
||||
println(sys.argv)
|
||||
pp.pipeline_mp_inp_dir(inp_root, n_p)
|
||||
println("end preprocess")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
preprocess_trainset(inp_root, sr, n_p, exp_dir)
|
||||
|
||||
@@ -27,6 +27,18 @@ def cache_harvest_f0(input_audio_path, fs, f0max, f0min, frame_period):
|
||||
f0 = pyworld.stonemask(audio, f0, t, fs)
|
||||
return f0
|
||||
|
||||
@lru_cache
|
||||
def cache_dio_f0(input_audio_path, fs, f0max, f0min, frame_period):
|
||||
audio = input_audio_path2wav[input_audio_path]
|
||||
f0, t = pyworld.dio(
|
||||
audio,
|
||||
fs=fs,
|
||||
f0_ceil=f0max,
|
||||
f0_floor=f0min,
|
||||
frame_period=frame_period,
|
||||
)
|
||||
f0 = pyworld.stonemask(audio, f0, t, fs)
|
||||
return f0
|
||||
|
||||
def change_rms(data1, sr1, data2, sr2, rate): # 1是输入音频,2是输出音频,rate是2的占比
|
||||
# print(data1.max(),data2.max())
|
||||
@@ -72,6 +84,7 @@ class VC(object):
|
||||
def get_f0(
|
||||
self,
|
||||
input_audio_path,
|
||||
rmvpe_model_path,
|
||||
x,
|
||||
p_len,
|
||||
f0_up_key,
|
||||
@@ -106,6 +119,11 @@ class VC(object):
|
||||
f0 = cache_harvest_f0(input_audio_path, self.sr, f0_max, f0_min, 10)
|
||||
if filter_radius > 2:
|
||||
f0 = signal.medfilt(f0, 3)
|
||||
elif f0_method == "dio":
|
||||
input_audio_path2wav[input_audio_path] = x.astype(np.double)
|
||||
f0 = cache_dio_f0(input_audio_path, self.sr, f0_max, f0_min, 10)
|
||||
if filter_radius > 2:
|
||||
f0 = signal.medfilt(f0, 3)
|
||||
elif f0_method == "crepe":
|
||||
model = "full"
|
||||
# Pick a batch size that doesn't cause memory errors on your gpu
|
||||
@@ -132,7 +150,8 @@ class VC(object):
|
||||
from lib.rmvpe import RMVPE
|
||||
print("loading rmvpe model")
|
||||
self.model_rmvpe = RMVPE(
|
||||
"rmvpe.pt", is_half=self.is_half, device=self.device
|
||||
#"rmvpe.pt", is_half=self.is_half, device=self.device
|
||||
rmvpe_model_path, is_half=self.is_half, device=self.device
|
||||
)
|
||||
|
||||
f0 = self.model_rmvpe.infer_from_audio(x, thred=0.03)
|
||||
@@ -275,6 +294,7 @@ class VC(object):
|
||||
sid,
|
||||
audio,
|
||||
input_audio_path,
|
||||
rmvpe_model_path,
|
||||
times,
|
||||
f0_up_key,
|
||||
f0_method,
|
||||
@@ -344,6 +364,7 @@ class VC(object):
|
||||
if if_f0 == 1:
|
||||
pitch, pitchf = self.get_f0(
|
||||
input_audio_path,
|
||||
rmvpe_model_path,
|
||||
audio_pad,
|
||||
p_len,
|
||||
f0_up_key,
|
||||
|
||||
Reference in New Issue
Block a user