RVC-Project
diff --git a/‎MDXNet.py
Lines changed: 15 additions & 8 deletions b/‎MDXNet.py
Lines changed: 15 additions & 8 deletions
diff --git a/‎config.py
Lines changed: 28 additions & 8 deletions b/‎config.py
Lines changed: 28 additions & 8 deletions
diff --git a/‎extract_f0_rmvpe.py
Lines changed: 2 additions & 2 deletions b/‎extract_f0_rmvpe.py
Lines changed: 2 additions & 2 deletions
diff --git a/‎extract_f0_rmvpe_dml.py
Lines changed: 129 additions & 0 deletions b/‎extract_f0_rmvpe_dml.py
Lines changed: 129 additions & 0 deletions
diff --git a/‎extract_feature_print.py
Lines changed: 18 additions & 9 deletions b/‎extract_feature_print.py
Lines changed: 18 additions & 9 deletions
diff --git a/‎gui_v1.py
Lines changed: 14 additions & 14 deletions b/‎gui_v1.py
Lines changed: 14 additions & 14 deletions
@@ -1,7 +1,6 @@
 import soundfile as sf
 import torch, pdb, os, warnings, librosa
 import numpy as np
-import onnxruntime as ort
 from tqdm import tqdm
 import torch
 
@@ -83,13 +82,19 @@ def get_models(device, dim_f, dim_t, n_fft):
 
 
 warnings.filterwarnings("ignore")
+import sys
+now_dir = os.getcwd()
+sys.path.append(now_dir)
+from config import Config
+
 cpu = torch.device("cpu")
-if torch.cuda.is_available():
-    device = torch.device("cuda:0")
-elif torch.backends.mps.is_available():
-    device = torch.device("mps")
-else:
-    device = torch.device("cpu")
+device=Config().device
+# if torch.cuda.is_available():
+#     device = torch.device("cuda:0")
+# elif torch.backends.mps.is_available():
+#     device = torch.device("mps")
+# else:
+#     device = torch.device("cpu")
 
 
 class Predictor:
@@ -98,9 +103,11 @@ def __init__(self, args):
         self.model_ = get_models(
             device=cpu, dim_f=args.dim_f, dim_t=args.dim_t, n_fft=args.n_fft
         )
+        import onnxruntime as ort
+        print(ort.get_available_providers())
         self.model = ort.InferenceSession(
             os.path.join(args.onnx, self.model_.target_name + ".onnx"),
-            providers=["CUDAExecutionProvider", "CPUExecutionProvider"],
+            providers=["CUDAExecutionProvider", "DmlExecutionProvider","CPUExecutionProvider"],
         )
         print("onnx load done")
 
 
@@ -36,10 +36,10 @@ def __init__(self):
             self.noparallel,
             self.noautoopen,
         ) = self.arg_parse()
+        self.instead=""
         self.x_pad, self.x_query, self.x_center, self.x_max = self.device_config()
 
-    @staticmethod
-    def arg_parse() -> tuple:
+    def arg_parse(self) -> tuple:
         exe = sys.executable or "python"
         parser = argparse.ArgumentParser()
         parser.add_argument("--port", type=int, default=7865, help="Listen port")
@@ -53,10 +53,15 @@ def arg_parse() -> tuple:
             action="store_true",
             help="Do not open in browser automatically",
         )
+        parser.add_argument(
+            "--dml",
+            action="store_true",
+            help="torch_dml",
+        )
         cmd_opts = parser.parse_args()
 
         cmd_opts.port = cmd_opts.port if 0 <= cmd_opts.port <= 65535 else 7865
-
+        self.dml=cmd_opts.dml
         return (
             cmd_opts.pycmd,
             cmd_opts.port,
@@ -106,13 +111,13 @@ def device_config(self) -> tuple:
                 with open("trainset_preprocess_pipeline_print.py", "w") as f:
                     f.write(strr)
         elif self.has_mps():
-            print("No supported Nvidia GPU found, use MPS instead")
-            self.device = "mps"
+            print("No supported Nvidia GPU found")
+            self.device = self.instead="mps"
             self.is_half = False
             use_fp32_config()
         else:
-            print("No supported Nvidia GPU found, use CPU instead")
-            self.device = "cpu"
+            print("No supported Nvidia GPU found")
+            self.device = self.instead="cpu"
             self.is_half = False
             use_fp32_config()
 
@@ -137,5 +142,20 @@ def device_config(self) -> tuple:
             x_query = 5
             x_center = 30
             x_max = 32
-
+        if(self.dml==True):
+            print("use DirectML instead")
+            try:os.rename("runtime\Lib\site-packages\onnxruntime","runtime\Lib\site-packages\onnxruntime-cuda")
+            except:pass
+            try:os.rename("runtime\Lib\site-packages\onnxruntime-dml","runtime\Lib\site-packages\onnxruntime")
+            except:pass
+            import torch_directml
+            self.device= torch_directml.device(torch_directml.default_device())
+            self.is_half=False
+        else:
+            if(self.instead):
+                print("use %s instead"%self.instead)
+            try:os.rename("runtime\Lib\site-packages\onnxruntime","runtime\Lib\site-packages\onnxruntime-cuda")
+            except:pass
+            try:os.rename("runtime\Lib\site-packages\onnxruntime-dml","runtime\Lib\site-packages\onnxruntime")
+            except:pass
         return x_pad, x_query, x_center, x_max
@@ -36,13 +36,13 @@ def __init__(self, samplerate=16000, hop_size=160):
 
     def compute_f0(self, path, f0_method):
         x = load_audio(path, self.fs)
-        p_len = x.shape[0] // self.hop
+        # p_len = x.shape[0] // self.hop
         if f0_method == "rmvpe":
             if hasattr(self, "model_rmvpe") == False:
                 from lib.rmvpe import RMVPE
 
                 print("loading rmvpe model")
-                self.model_rmvpe = RMVPE("rmvpe.pt", is_half=True, device="cuda")
+                self.model_rmvpe = RMVPE("rmvpe.pt", is_half=is_half, device="cuda")
             f0 = self.model_rmvpe.infer_from_audio(x, thred=0.03)
         return f0
 
 
@@ -0,0 +1,129 @@
+import os, traceback, sys, parselmouth
+
+now_dir = os.getcwd()
+sys.path.append(now_dir)
+from lib.audio import load_audio
+import pyworld
+import numpy as np, logging
+
+logging.getLogger("numba").setLevel(logging.WARNING)
+
+exp_dir = sys.argv[1]
+import torch_directml
+device = torch_directml.device(torch_directml.default_device())
+f = open("%s/extract_f0_feature.log" % exp_dir, "a+")
+
+
+def printt(strr):
+    print(strr)
+    f.write("%s\n" % strr)
+    f.flush()
+
+
+class FeatureInput(object):
+    def __init__(self, samplerate=16000, hop_size=160):
+        self.fs = samplerate
+        self.hop = hop_size
+
+        self.f0_bin = 256
+        self.f0_max = 1100.0
+        self.f0_min = 50.0
+        self.f0_mel_min = 1127 * np.log(1 + self.f0_min / 700)
+        self.f0_mel_max = 1127 * np.log(1 + self.f0_max / 700)
+
+    def compute_f0(self, path, f0_method):
+        x = load_audio(path, self.fs)
+        # p_len = x.shape[0] // self.hop
+        if f0_method == "rmvpe":
+            if hasattr(self, "model_rmvpe") == False:
+                from lib.rmvpe import RMVPE
+
+                print("loading rmvpe model")
+                self.model_rmvpe = RMVPE("rmvpe.pt", is_half=False, device=device)
+            f0 = self.model_rmvpe.infer_from_audio(x, thred=0.03)
+        return f0
+
+    def coarse_f0(self, f0):
+        f0_mel = 1127 * np.log(1 + f0 / 700)
+        f0_mel[f0_mel > 0] = (f0_mel[f0_mel > 0] - self.f0_mel_min) * (
+            self.f0_bin - 2
+        ) / (self.f0_mel_max - self.f0_mel_min) + 1
+
+        # use 0 or 1
+        f0_mel[f0_mel <= 1] = 1
+        f0_mel[f0_mel > self.f0_bin - 1] = self.f0_bin - 1
+        f0_coarse = np.rint(f0_mel).astype(int)
+        assert f0_coarse.max() <= 255 and f0_coarse.min() >= 1, (
+            f0_coarse.max(),
+            f0_coarse.min(),
+        )
+        return f0_coarse
+
+    def go(self, paths, f0_method):
+        if len(paths) == 0:
+            printt("no-f0-todo")
+        else:
+            printt("todo-f0-%s" % len(paths))
+            n = max(len(paths) // 5, 1)  # 每个进程最多打印5条
+            for idx, (inp_path, opt_path1, opt_path2) in enumerate(paths):
+                try:
+                    if idx % n == 0:
+                        printt("f0ing,now-%s,all-%s,-%s" % (idx, len(paths), inp_path))
+                    if (
+                        os.path.exists(opt_path1 + ".npy") == True
+                        and os.path.exists(opt_path2 + ".npy") == True
+                    ):
+                        continue
+                    featur_pit = self.compute_f0(inp_path, f0_method)
+                    np.save(
+                        opt_path2,
+                        featur_pit,
+                        allow_pickle=False,
+                    )  # nsf
+                    coarse_pit = self.coarse_f0(featur_pit)
+                    np.save(
+                        opt_path1,
+                        coarse_pit,
+                        allow_pickle=False,
+                    )  # ori
+                except:
+                    printt("f0fail-%s-%s-%s" % (idx, inp_path, traceback.format_exc()))
+
+
+if __name__ == "__main__":
+    # exp_dir=r"E:\codes\py39\dataset\mi-test"
+    # n_p=16
+    # f = open("%s/log_extract_f0.log"%exp_dir, "w")
+    printt(sys.argv)
+    featureInput = FeatureInput()
+    paths = []
+    inp_root = "%s/1_16k_wavs" % (exp_dir)
+    opt_root1 = "%s/2a_f0" % (exp_dir)
+    opt_root2 = "%s/2b-f0nsf" % (exp_dir)
+
+    os.makedirs(opt_root1, exist_ok=True)
+    os.makedirs(opt_root2, exist_ok=True)
+    for name in sorted(list(os.listdir(inp_root))):
+        inp_path = "%s/%s" % (inp_root, name)
+        if "spec" in inp_path:
+            continue
+        opt_path1 = "%s/%s" % (opt_root1, name)
+        opt_path2 = "%s/%s" % (opt_root2, name)
+        paths.append([inp_path, opt_path1, opt_path2])
+    try:
+        featureInput.go(paths, "rmvpe")
+    except:
+        printt("f0_all_fail-%s" % (traceback.format_exc()))
+    # ps = []
+    # for i in range(n_p):
+    #     p = Process(
+    #         target=featureInput.go,
+    #         args=(
+    #             paths[i::n_p],
+    #             f0method,
+    #         ),
+    #     )
+    #     ps.append(p)
+    #     p.start()
+    # for i in range(n_p):
+    #     ps[i].join()
@@ -3,7 +3,7 @@
 os.environ["PYTORCH_ENABLE_MPS_FALLBACK"] = "1"
 os.environ["PYTORCH_MPS_HIGH_WATERMARK_RATIO"] = "0.0"
 
-# device=sys.argv[1]
+device=sys.argv[1]
 n_part = int(sys.argv[2])
 i_part = int(sys.argv[3])
 if len(sys.argv) == 6:
@@ -18,13 +18,22 @@
 import torch.nn.functional as F
 import soundfile as sf
 import numpy as np
-from fairseq import checkpoint_utils
-
-device = "cpu"
-if torch.cuda.is_available():
-    device = "cuda"
-elif torch.backends.mps.is_available():
-    device = "mps"
+import fairseq
+
+if("privateuseone"not in device):
+    device = "cpu"
+    if torch.cuda.is_available():
+        device = "cuda"
+    elif torch.backends.mps.is_available():
+        device = "mps"
+else:
+    import torch_directml
+    device = torch_directml.device(torch_directml.default_device())
+    def forward_dml(ctx, x, scale):
+        ctx.scale = scale
+        res = x.clone().detach()
+        return res
+    fairseq.modules.grad_multiply.GradMultiply.forward=forward_dml
 
 f = open("%s/extract_f0_feature.log" % exp_dir, "a+")
 
@@ -70,7 +79,7 @@ def readwave(wav_path, normalize=False):
         % model_path
     )
     exit(0)
-models, saved_cfg, task = checkpoint_utils.load_model_ensemble_and_task(
+models, saved_cfg, task = fairseq.checkpoint_utils.load_model_ensemble_and_task(
     [model_path],
     suffix="",
 )
 
@@ -1,5 +1,5 @@
-import os, sys
-
+import os, sys,pdb
+os.environ["OMP_NUM_THREADS"]="2"
 if sys.platform == "darwin":
     os.environ["PYTORCH_ENABLE_MPS_FALLBACK"] = "1"
 
@@ -46,20 +46,21 @@ def run(self):
     import torch.nn.functional as F
     import torchaudio.transforms as tat
     from i18n import I18nAuto
-
+    import rvc_for_realtime
     i18n = I18nAuto()
-    device = torch.device(
-        "cuda"
-        if torch.cuda.is_available()
-        else ("mps" if torch.backends.mps.is_available() else "cpu")
-    )
+    device=rvc_for_realtime.config.device
+    # device = torch.device(
+    #     "cuda"
+    #     if torch.cuda.is_available()
+    #     else ("mps" if torch.backends.mps.is_available() else "cpu")
+    # )
     current_dir = os.getcwd()
     inp_q = Queue()
     opt_q = Queue()
     n_cpu = min(cpu_count(), 8)
     for _ in range(n_cpu):
         Harvest(inp_q, opt_q).start()
-    from rvc_for_realtime import RVC
+
 
     class GUIConfig:
         def __init__(self) -> None:
@@ -75,7 +76,7 @@ def __init__(self) -> None:
             self.I_noise_reduce = False
             self.O_noise_reduce = False
             self.index_rate = 0.3
-            self.n_cpu = min(n_cpu, 8)
+            self.n_cpu = min(n_cpu, 6)
             self.f0method = "harvest"
             self.sg_input_device = ""
             self.sg_output_device = ""
@@ -239,7 +240,7 @@ def launcher(self):
                             [
                                 sg.Text(i18n("采样长度")),
                                 sg.Slider(
-                                    range=(0.12, 2.4),
+                                    range=(0.09, 2.4),
                                     key="block_time",
                                     resolution=0.03,
                                     orientation="h",
@@ -271,7 +272,7 @@ def launcher(self):
                             [
                                 sg.Text(i18n("额外推理时长")),
                                 sg.Slider(
-                                    range=(0.05, 3.00),
+                                    range=(0.05, 5.00),
                                     key="extra_time",
                                     resolution=0.01,
                                     orientation="h",
@@ -391,7 +392,7 @@ def set_values(self, values):
         def start_vc(self):
             torch.cuda.empty_cache()
             self.flag_vc = True
-            self.rvc = RVC(
+            self.rvc = rvc_for_realtime.RVC(
                 self.config.pitch,
                 self.config.pth_path,
                 self.config.index_path,
@@ -510,7 +511,6 @@ def audio_callback(
             self.input_wav[:] = np.append(self.input_wav[self.block_frame :], indata)
             # infer
             inp = torch.from_numpy(self.input_wav).to(device)
-            ##0
             res1 = self.resampler(inp)
             ###55%
             rate1 = self.block_frame / (