Spaces:

JotunnBurton
/

wuwa-bert-vits2

Sleeping

App Files Files Community

JotunnBurton commited on Apr 16

Commit

7709d54

verified ·

1 Parent(s): c069afe

Update app.py

Browse files

Files changed (1) hide show

app.py +64 -9

app.py CHANGED Viewed

@@ -50,26 +50,72 @@ def get_text(text, hps):
     return bert, ja_bert, phone, tone, language
-def infer(text, sdp_ratio, noise_scale, noise_scale_w, length_scale, sid, net_g_ms, hps):
-    bert, ja_bert, phones, tones, lang_ids = get_text(text, hps)
     with torch.no_grad():
         x_tst = phones.to(device).unsqueeze(0)
         tones = tones.to(device).unsqueeze(0)
         lang_ids = lang_ids.to(device).unsqueeze(0)
         bert = bert.to(device).unsqueeze(0)
-        ja_bert = ja_bert.to(device).unsqueeze(0)
         x_tst_lengths = torch.LongTensor([phones.size(0)]).to(device)
         del phones
-        sid = torch.LongTensor([sid]).to(device)
         audio = (
-            net_g_ms.infer(
                 x_tst,
                 x_tst_lengths,
-                sid,
                 tones,
                 lang_ids,
                 bert,
-                ja_bert,
                 sdp_ratio=sdp_ratio,
                 noise_scale=noise_scale,
                 noise_scale_w=noise_scale_w,
@@ -79,8 +125,17 @@ def infer(text, sdp_ratio, noise_scale, noise_scale_w, length_scale, sid, net_g_
             .float()
             .numpy()
         )
-        del x_tst, tones, lang_ids, bert, x_tst_lengths, sid
-        torch.cuda.empty_cache()
         return audio

     return bert, ja_bert, phone, tone, language
+def infer(
+    text,
+    sdp_ratio,
+    noise_scale,
+    noise_scale_w,
+    length_scale,
+    sid,
+    language,
+    hps,
+    net_g,
+    device,
+    emotion,
+    reference_audio=None,
+    skip_start=False,
+    skip_end=False,
+    style_text=None,
+    style_weight=0.7,
+    text_mode="Text",
+):
+    # 2.2版本参数位置变了
+    # 2.1 参数新增 emotion reference_audio skip_start skip_end
+    version = hps.version if hasattr(hps, "version") else latest_version
+    language = "JP"
+    if isinstance(reference_audio, np.ndarray):
+        emo = get_clap_audio_feature(reference_audio, device)
+    else:
+        emo = get_clap_text_feature(emotion, device)
+    emo = torch.squeeze(emo, dim=1)
+    bert, phones, tones, lang_ids = get_text(
+        text,
+        language,
+        hps,
+        device,
+        style_text=style_text,
+        style_weight=style_weight,
+    )
+    if skip_start:
+        phones = phones[3:]
+        tones = tones[3:]
+        lang_ids = lang_ids[3:]
+        bert = bert[:, 3:]
+    if skip_end:
+        phones = phones[:-2]
+        tones = tones[:-2]
+        lang_ids = lang_ids[:-2]
+        bert = bert[:, :-2]
     with torch.no_grad():
         x_tst = phones.to(device).unsqueeze(0)
         tones = tones.to(device).unsqueeze(0)
         lang_ids = lang_ids.to(device).unsqueeze(0)
         bert = bert.to(device).unsqueeze(0)
         x_tst_lengths = torch.LongTensor([phones.size(0)]).to(device)
+        emo = emo.to(device).unsqueeze(0)
         del phones
+        speakers = torch.LongTensor([hps.data.spk2id[sid]]).to(device)
+        print(text)
         audio = (
+            net_g.infer(
                 x_tst,
                 x_tst_lengths,
+                speakers,
                 tones,
                 lang_ids,
                 bert,
+                emo,
                 sdp_ratio=sdp_ratio,
                 noise_scale=noise_scale,
                 noise_scale_w=noise_scale_w,
             .float()
             .numpy()
         )
+        del (
+            x_tst,
+            tones,
+            lang_ids,
+            bert,
+            x_tst_lengths,
+            speakers,
+            emo,
+        )  # , emo
+        if torch.cuda.is_available():
+            torch.cuda.empty_cache()
         return audio