sush0401 commited on
Commit
c169519
·
verified ·
1 Parent(s): 58edc21

DreamVoice: ZeroGPU app

Browse files
Files changed (1) hide show
  1. tts.py +125 -0
tts.py ADDED
@@ -0,0 +1,125 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """VoxCPM2 voice cloning and narration — local GPU inference for HF Spaces."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ import tempfile
8
+
9
+ import numpy as np
10
+
11
+ MODEL_ID = "openbmb/VoxCPM2"
12
+
13
+ _model = None
14
+
15
+
16
+ def _get_model():
17
+ global _model
18
+ if _model is None:
19
+ from voxcpm import VoxCPM
20
+ _model = VoxCPM.from_pretrained(
21
+ MODEL_ID,
22
+ device="cuda",
23
+ load_denoiser=True,
24
+ optimize=True,
25
+ )
26
+ return _model
27
+
28
+
29
+ # Load at module level for ZeroGPU (CUDA emulation outside @spaces.GPU)
30
+ try:
31
+ _get_model()
32
+ except Exception:
33
+ pass
34
+
35
+
36
+ def _split_sentences(text: str, max_chars: int = 200):
37
+ parts = re.split(r"(?<=[.!?।])\s+|\n+", text.strip())
38
+ out = []
39
+ for p in parts:
40
+ p = p.strip()
41
+ if not p:
42
+ continue
43
+ while len(p) > max_chars:
44
+ cut = p.rfind(" ", 0, max_chars)
45
+ cut = cut if cut > 0 else max_chars
46
+ out.append(p[:cut].strip())
47
+ p = p[cut:].strip()
48
+ out.append(p)
49
+ return out or [text.strip()]
50
+
51
+
52
+ def _pause_for(mood: str, energy: float = 0.45) -> float:
53
+ energy = max(0.0, min(1.0, float(energy)))
54
+ base = 0.45 if mood in ("funny", "magical") else 0.65
55
+ return round(base + (0.85 - base) * (1.0 - energy), 3)
56
+
57
+
58
+ def _with_bedtime_style(text: str, speed: float, mood: str = "", energy: float = 0.45) -> str:
59
+ energy = max(0.0, min(1.0, float(energy)))
60
+ mood_styles = {
61
+ "magical": "gentle, warm, slightly slow, wonder-filled whisper, like telling a secret about something beautiful, soft rising intonation on wonder words, pause briefly after each beat",
62
+ "funny": "warm, playful, slightly animated, gentle humor in the voice, light chuckle between lines, bright and cheerful but still soft enough for bedtime, slightly faster pace than usual",
63
+ "calming": "very slow, deep warm whisper, barely above a breath, each word drifting gently into the next, long pauses between sentences, voice fading softly at the end of each line, like someone falling asleep while reading",
64
+ "dreamy": "slow, soft, breathy whisper, voice drifting like floating on a cloud, elongated vowels, gentle hum between phrases, lullaby-like rhythm, words dissolving into silence",
65
+ }
66
+ style = mood_styles.get(mood, "gentle, warm, sleepy bedtime voice, slightly slow pace")
67
+ if energy >= 0.66:
68
+ style += ", a little brighter and more animated, lively for a delighted child"
69
+ elif energy <= 0.33:
70
+ style += ", even softer and slower, barely above a whisper"
71
+ return f"({style}){text}"
72
+
73
+
74
+ def _postprocess_np(audio, sr):
75
+ from audio_postprocess import postprocess
76
+ return postprocess(audio, sr)
77
+
78
+
79
+ def clone_and_speak(ref_wav: str, text: str, speed: float = 0.9, mood: str = "", energy: float = 0.45) -> str:
80
+ """Clone the reference voice and synthesize text to a temporary WAV path.
81
+
82
+ Runs on local GPU (HF Spaces GPU Zero).
83
+ """
84
+ if not ref_wav or not os.path.exists(ref_wav):
85
+ raise ValueError("Please provide a prepared voice reference WAV.")
86
+
87
+ story_text = (text or "").strip()
88
+ if not story_text:
89
+ raise ValueError("Please provide story text to narrate.")
90
+
91
+ model = _get_model()
92
+ sr = int(model.tts_model.sample_rate)
93
+
94
+ pause = _pause_for(mood, energy)
95
+ silence = np.zeros(int(pause * sr), dtype=np.float32)
96
+
97
+ chunks = []
98
+ for sentence in _split_sentences(story_text):
99
+ wav = model.generate(
100
+ text=_with_bedtime_style(sentence, speed, mood, energy),
101
+ reference_wav_path=ref_wav,
102
+ cfg_value=2.0,
103
+ inference_timesteps=10,
104
+ normalize=True,
105
+ denoise=True,
106
+ retry_badcase=True,
107
+ retry_badcase_max_times=3,
108
+ retry_badcase_ratio_threshold=8.0,
109
+ )
110
+ wav = np.asarray(wav, dtype=np.float32)
111
+ if wav.size:
112
+ chunks.append(wav)
113
+ chunks.append(silence)
114
+
115
+ if not chunks:
116
+ raise RuntimeError("VoxCPM2 produced no audio.")
117
+
118
+ full = np.concatenate(chunks)
119
+ full = _postprocess_np(full, sr)
120
+
121
+ import soundfile as sf
122
+ fd, out_path = tempfile.mkstemp(prefix="dreamvoice_story_", suffix=".wav")
123
+ os.close(fd)
124
+ sf.write(out_path, full, sr)
125
+ return out_path