-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathui-clone.py
More file actions
190 lines (154 loc) · 8.55 KB
/
Copy pathui-clone.py
File metadata and controls
190 lines (154 loc) · 8.55 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
#!/usr/bin/env python3
"""Simple Gradio UI for qwen-tts via clone.sh"""
import os
import subprocess
import tempfile
from pathlib import Path
from dotenv import load_dotenv
import gradio as gr
# ── load .env if present ─────────────────────────────────────────────
load_dotenv(Path(__file__).parent / ".env")
SCRIPT = str(Path(__file__).parent / "clone.sh")
OUTPUT_WAV = "/tmp/out.wav"
EMBEDDINGS_DIR = Path(__file__).parent / "embeddings"
def synthesize(ref_wav: str, ref_text_path: str, prompt: str, lang: str,
ref_spk: str, ref_rvq: str, ref_text_override: str,
seed: int, greedy: bool, temp: float, top_k: int, top_p: float,
rep_pen: float, sub_temp: float, fmt: str, max_new: int,
stream: bool, no_fa: bool, clamp_fp16: bool, output_path: str):
"""Run clone.sh with user-provided inputs."""
if not (Path(__file__).parent / ".env").exists():
raise gr.Error("Missing .env — copy .env.example to .env and configure it (see README.md).")
prompt_path = tempfile.NamedTemporaryFile(
mode="w", suffix=".txt", delete=False
).name
Path(prompt_path).write_text(prompt)
env = os.environ.copy()
if seed != -1:
env["QWENTTS_SEED"] = str(seed)
if greedy:
env["QWENTTS_GREEDY"] = "1"
env["QWENTTS_TEMP"] = str(temp)
env["QWENTTS_TOP_K"] = str(top_k)
env["QWENTTS_TOP_P"] = str(top_p)
env["QWENTTS_REP_PEN"] = str(rep_pen)
env["QWENTTS_SUB_TEMP"] = str(sub_temp)
env["QWENTTS_FORMAT"] = fmt
env["QWENTTS_MAX_NEW"] = str(max_new)
if stream:
env["QWENTTS_STREAM"] = "1"
if no_fa:
env["QWENTTS_NO_FA"] = "1"
if clamp_fp16:
env["QWENTTS_CLAMP_FP16"] = "1"
if output_path:
env["QWENTTS_OUTPUT"] = output_path
# Pre-encoded reference overrides raw WAV
if ref_spk:
env["QWENTTS_REF_SPK"] = ref_spk
if ref_rvq:
env["QWENTTS_REF_RVQ"] = ref_rvq
if ref_text_override:
env["QWENTTS_REF_TEXT"] = ref_text_override
try:
subprocess.run(
["bash", SCRIPT, ref_wav or "", ref_text_path or "", prompt_path, lang],
check=True, env=env,
)
return env.get("QWENTTS_OUTPUT", OUTPUT_WAV)
except subprocess.CalledProcessError as e:
raise gr.Error(f"clone.sh failed (exit {e.returncode}). Check terminal for details.")
finally:
Path(prompt_path).unlink(missing_ok=True)
# ── helpers ──────────────────────────────────────────────────────────
_LANGS = ["French", "English", "Spanish", "German", "Italian", "Portuguese",
"Chinese", "Japanese", "Korean", "Russian"]
_DEFAULT_LANG = os.environ.get("LANG", "French") if os.environ.get("LANG", "") in _LANGS else "French"
_PRESETS = [p.stem for p in sorted(EMBEDDINGS_DIR.glob("*.spk"))]
def _build_advanced():
"""Return shared advanced-options widgets as a list."""
with gr.Accordion("Advanced Options", open=False):
seed = gr.Number(label="Seed (-1 = random)", value=-1, precision=0)
greedy = gr.Checkbox(label="Greedy decoding", value=False)
temp = gr.Slider(label="Temperature", minimum=0.0, maximum=2.0, step=0.01, value=0.7)
top_k = gr.Number(label="Top-k", value=40, precision=0)
top_p = gr.Slider(label="Top-p", minimum=0.0, maximum=1.0, step=0.01, value=0.9)
rep_pen = gr.Slider(label="Repetition Penalty", minimum=1.0, maximum=2.0, step=0.01, value=1.1)
sub_temp = gr.Slider(label="Sub-quantizer Temp", minimum=0.0, maximum=1.0, step=0.01, value=0.1)
fmt = gr.Dropdown(label="Output Format", choices=["wav16", "wav24", "wav32"], value="wav16")
max_new = gr.Number(label="Max New Tokens", value=4096, precision=0)
stream = gr.Checkbox(label="Stream by line", value=False)
no_fa = gr.Checkbox(label="Disable Flash Attention", value=False)
clamp_fp16 = gr.Checkbox(label="Clamp FP16", value=False)
output_path = gr.Textbox(label="Output Path (optional)", placeholder="/tmp/out.wav")
return [seed, greedy, temp, top_k, top_p, rep_pen, sub_temp,
fmt, max_new, stream, no_fa, clamp_fp16, output_path]
def _build_synthesize(ref_wav, ref_text, prompt, lang,
spk_path, rvq_path, ref_text_override, *adv):
"""Thin wrapper so both tabs can reuse synthesize."""
seed, greedy, temp, top_k, top_p, rep_pen, sub_temp, \
fmt, max_new, stream, no_fa, clamp_fp16, output_path = adv
return synthesize(ref_wav, ref_text, prompt, lang,
spk_path, rvq_path, ref_text_override,
seed, greedy, temp, top_k, top_p,
rep_pen, sub_temp, fmt, max_new,
stream, no_fa, clamp_fp16, output_path)
# ── UI ───────────────────────────────────────────────────────────────
with gr.Blocks(title="Qwen TTS - Clone") as demo:
gr.Markdown("# Qwen TTS - Clone")
with gr.Tabs():
# ── Tab 1: WAV + Text ────────────────────────────────────
with gr.Tab("Clone (WAV)"):
gr.Markdown("Upload a reference WAV and its transcript text.")
with gr.Row():
ref_wav = gr.Audio(label="Reference WAV", type="filepath",
sources=["upload"], buttons=["download"])
ref_text = gr.File(label="Reference Text File", file_types=[".txt"])
prompt1 = gr.Textbox(label="Prompt Text", lines=6,
placeholder="Enter the text you want spoken…")
lang1 = gr.Dropdown(choices=_LANGS, value=_DEFAULT_LANG, label="Language")
adv1 = _build_advanced()
btn1 = gr.Button("Synthesize", variant="primary")
stop1 = gr.Button("Stop", variant="stop")
out1 = gr.Audio(label="Output Audio", type="filepath", buttons=["download"])
inputs1 = [ref_wav, ref_text, prompt1, lang1,
gr.State(None), gr.State(None), gr.State("")] + adv1
task1 = btn1.click(_build_synthesize, inputs=inputs1, outputs=out1)
stop1.click(None, None, None, cancels=[task1])
# ── Tab 2: Embeddings ────────────────────────────────────
with gr.Tab("Embeddings"):
gr.Markdown(
"Use pre-encoded speaker embeddings (`.spk` + `.rvq`) instead of WAV+text."
)
with gr.Row():
preset = gr.Dropdown(
choices=_PRESETS,
value=_PRESETS[0] if _PRESETS else None,
label="Preset Voice",
interactive=True,
)
preview_btn = gr.Button("🔊 Preview", variant="secondary")
emb_ref_text = gr.File(label="Reference Text File", file_types=[".txt"])
prompt2 = gr.Textbox(label="Prompt Text", lines=6,
placeholder="Enter the text you want spoken…")
lang2 = gr.Dropdown(choices=_LANGS, value=_DEFAULT_LANG, label="Language")
adv2 = _build_advanced()
btn2 = gr.Button("Synthesize", variant="primary")
stop2 = gr.Button("Stop", variant="stop")
out2 = gr.Audio(label="Output Audio", type="filepath", buttons=["download"])
def _preview(v):
if v:
wav = str(EMBEDDINGS_DIR / f"{v}.wav")
if Path(wav).exists():
return wav
return None
preview_btn.click(_preview, inputs=preset, outputs=out2)
def _emb_synthesize(ref_text_val, prompt_text, lang_val, voice, *adv_vals):
spk_path = str(EMBEDDINGS_DIR / f"{voice}.spk") if voice else ""
rvq_path = str(EMBEDDINGS_DIR / f"{voice}.rvq") if voice else ""
return synthesize("", "", prompt_text, lang_val,
spk_path, rvq_path, ref_text_val or "", *adv_vals)
inputs2 = [emb_ref_text, prompt2, lang2, preset] + adv2
task2 = btn2.click(_emb_synthesize, inputs=inputs2, outputs=out2)
stop2.click(None, None, None, cancels=[task2])
demo.launch(server_name="127.0.0.1", server_port=7860, share=False)