llm67 / app.py
humair025's picture
Update app.py
6f9a31e verified
Raw History Blame Contribute Delete
22.1 kB
"""
app.py β€” Gradio frontend for Neural TTS
Transformers / ONNX / GGUF LLM β†’ ONNX Decoder + Vocoder (humair025/devo)
LLM: zuhri025/tts_weights_text_only_soprano
Supports CUDA, MPS, CPU.
"""
import os
os.environ["OMP_NUM_THREADS"] = "2"
os.environ["MKL_NUM_THREADS"] = "2"
os.environ["OPENBLAS_NUM_THREADS"] = "2"
os.environ["VECLIB_MAXIMUM_THREADS"] = "2"
os.environ["NUMEXPR_NUM_THREADS"] = "2"
import tempfile
import numpy as np
import soundfile as sf
import torch
torch.set_num_threads(2)
torch.set_num_interop_threads(2)
import gradio as gr
from tts_engine import (
text_to_audio,
load_style_embedding,
DEFAULT_LLM_REPO_ID,
DEFAULT_LLM_SUBFOLDER,
DEVO_REPO_ID,
DEFAULT_DECODER_FILE,
DEFAULT_VOCODER_FILE,
LLM_ONNX_REPO_ID,
LLM_ONNX_FILE,
LLM_GGUF_REPO_ID,
LLM_GGUF_FILE,
GGUF_QUANT_FILES,
SAMPLE_RATE,
_LLAMA_CPP_AVAILABLE,
)
# ─────────────────────────────────────────────────────────────────────────────
# DEVICE
# ─────────────────────────────────────────────────────────────────────────────
if torch.cuda.is_available():
_DEVICE_LABEL = f"🟒 GPU β€” {torch.cuda.get_device_name(0)}"
_USE_AUTOCAST = True
_PIN_MEMORY = True
_DEFAULT_GGUF_GPU_LAYERS = -1
elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
_DEVICE_LABEL = "🟑 MPS β€” Apple Silicon"
_USE_AUTOCAST = False
_PIN_MEMORY = False
_DEFAULT_GGUF_GPU_LAYERS = 0
else:
_DEVICE_LABEL = "πŸ”΅ CPU"
_USE_AUTOCAST = False
_PIN_MEMORY = False
_DEFAULT_GGUF_GPU_LAYERS = 0
# ─────────────────────────────────────────────────────────────────────────────
# VOICE PRESETS
# ─────────────────────────────────────────────────────────────────────────────
VOICES_HF_REPO = "humair025/voices"
VOICES_GENDERS = ["male", "female"]
_BASE_DIR = os.path.dirname(os.path.abspath(__file__))
VOICES_ROOT = os.path.join(_BASE_DIR, "voices")
def _hf_list_voice_files(gender: str) -> list:
try:
from huggingface_hub import list_repo_files
supported = (".pt", ".pth", ".npy")
return sorted(
os.path.basename(f)
for f in list_repo_files(VOICES_HF_REPO)
if f.startswith(f"{gender}/") and os.path.splitext(f)[1].lower() in supported
)
except Exception as e:
print(f"[voices] Could not list HF files for '{gender}': {e}")
return []
def _ensure_voice_file(gender: str, filename: str) -> str:
local_path = os.path.join(VOICES_ROOT, gender, filename)
if os.path.isfile(local_path):
return local_path
os.makedirs(os.path.join(VOICES_ROOT, gender), exist_ok=True)
try:
from huggingface_hub import hf_hub_download
return hf_hub_download(repo_id=VOICES_HF_REPO,
filename=f"{gender}/{filename}",
local_dir=VOICES_ROOT)
except Exception as e:
print(f"[voices] FAILED to download {gender}/{filename}: {e}")
return local_path
def _load_voices_for_gender(gender: str) -> dict:
voices = {}
supported = (".pt", ".pth", ".npy")
local_dir = os.path.join(VOICES_ROOT, gender)
local_files = set()
if os.path.isdir(local_dir):
local_files = {f for f in os.listdir(local_dir)
if os.path.splitext(f)[1].lower() in supported}
all_files = sorted(local_files | set(_hf_list_voice_files(gender)))
if not all_files:
print(f"[voices] ⚠️ No voice files for '{gender}'.")
return voices
for fname in all_files:
path = _ensure_voice_file(gender, fname)
if not os.path.isfile(path):
continue
try:
emb = load_style_embedding(path)
if emb is not None:
voices[os.path.splitext(fname)[0]] = emb
print(f"[voices] βœ… {fname} shape={emb.shape}")
except Exception as e:
print(f"[voices] SKIP {fname}: {e}")
return voices
print("[voices] Loading voice presets…")
PRESET_EMBEDDINGS: dict = {g: _load_voices_for_gender(g) for g in VOICES_GENDERS}
print("[voices] Ready β€” " + " ".join(
f"{g}: {len(v)} voice(s)" for g, v in PRESET_EMBEDDINGS.items()))
def _voices_for(gender: str) -> list:
return sorted(PRESET_EMBEDDINGS.get(gender, {}).keys())
def _default_voice(gender: str):
c = _voices_for(gender)
return c[0] if c else None
# ─────────────────────────────────────────────────────────────────────────────
# INFERENCE
# ─────────────────────────────────────────────────────────────────────────────
def generate_speech(
text, gender, preset_choice,
voice_tag_input, accent, mode,
lm_backend_type, gguf_quant, gguf_n_gpu_layers,
decoder_variant,
temperature, top_p, top_k, repetition_penalty, max_new_tokens, num_threads,
llm_repo, llm_subfolder, devo_repo,
progress=gr.Progress(track_tqdm=False),
):
if not text.strip():
raise gr.Error("Please enter some text to synthesise.")
# Style embedding
style_np = PRESET_EMBEDDINGS.get(gender, {}).get(preset_choice)
if style_np is None and PRESET_EMBEDDINGS.get(gender):
style_np = next(iter(PRESET_EMBEDDINGS[gender].values()))
# Normalise voice tag
vt = voice_tag_input.strip() if voice_tag_input else None
if vt and not vt.startswith("<|"):
vt = f"<|{vt}|>"
if not vt:
vt = None
# Decoder ONNX variant
decoder_file = {
"fp16 (fast, GPU)": "decoder_fp16.onnx",
"int8 (small, CPU)": "decoder_int8.onnx",
}.get(decoder_variant, DEFAULT_DECODER_FILE)
_be_label = {"transformers": "Transformers", "onnx": "ONNX LLM",
"gguf": f"GGUF ({gguf_quant})"}.get(lm_backend_type, lm_backend_type)
progress(0.1, desc=f"Preparing {_be_label} backend…")
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
out_path = f.name
try:
progress(0.2, desc="Generating speech…")
wav_np, sr, stats = text_to_audio(
text = text,
gender = gender,
voice_tag = vt,
accent = accent,
mode = mode,
style_path = style_np,
lm_backend_type = lm_backend_type,
llm_path = llm_repo,
llm_subfolder = llm_subfolder,
llm_onnx_repo = LLM_ONNX_REPO_ID,
llm_onnx_file = LLM_ONNX_FILE,
gguf_repo = LLM_GGUF_REPO_ID,
gguf_file = gguf_quant,
gguf_n_gpu_layers = int(gguf_n_gpu_layers),
gguf_n_ctx = 2048,
decoder_path = decoder_file,
vocoder_path = DEFAULT_VOCODER_FILE,
devo_repo_id = devo_repo,
device = "auto",
num_threads = int(num_threads),
save_path = out_path,
play = False,
verbose = True,
stream_generation = False,
use_autocast = _USE_AUTOCAST,
pin_memory = _PIN_MEMORY,
temperature = temperature,
top_p = top_p,
top_k = int(top_k),
repetition_penalty= repetition_penalty,
max_new_tokens = int(max_new_tokens),
)
progress(1.0, desc="Done!")
except Exception as e:
raise gr.Error(f"Generation failed: {e}")
rtf = stats.get("rtf_total", 0)
speed = stats.get("speed_x", 0)
dur = stats.get("audio_duration_s", 0)
t_tot = stats.get("t_total_s", 0)
t_lm = stats.get("t_lm_s", 0)
t_dec = stats.get("t_decode_s", 0)
info = (
f"**Preset:** {preset_choice or 'β€”'} ({gender}) &nbsp;|&nbsp; "
f"**Voice tag:** {vt or 'β€”'} &nbsp;|&nbsp; "
f"**Accent:** {accent} &nbsp;|&nbsp; "
f"**LM:** {stats.get('backend', '?')} &nbsp;|&nbsp; "
f"**Duration:** {dur:.2f}s &nbsp;|&nbsp; "
f"**Gen time:** {t_tot:.2f}s (LM {t_lm:.2f}s + decode {t_dec:.2f}s) &nbsp;|&nbsp; "
f"**Speed:** {speed:.2f}Γ— RT &nbsp;|&nbsp; "
f"**RTF:** {rtf:.4f}"
)
return out_path, info
# ─────────────────────────────────────────────────────────────────────────────
# UI HELPERS
# ─────────────────────────────────────────────────────────────────────────────
def _update_preset_dropdown(gender):
choices = _voices_for(gender)
return gr.update(choices=choices, value=choices[0] if choices else None)
def _update_gguf_visibility(lm_backend_type):
vis = lm_backend_type == "gguf"
return gr.update(visible=vis), gr.update(visible=vis)
# ─────────────────────────────────────────────────────────────────────────────
# CSS
# ─────────────────────────────────────────────────────────────────────────────
_CSS = """
@import url('https://fonts.googleapis.com/css2?family=Syne:wght@400;600;800&family=DM+Mono:ital,wght@0,400;0,500;1,400&display=swap');
:root {
--bg:#0b0d12; --surface:#13161f; --surface2:#1b1f2e;
--border:#1e293b; --border2:#334155;
--accent:#6ee7b7; --accent2:#38bdf8;
--text:#e2e8f0; --muted:#64748b;
--radius:12px; --fhead:'Syne',sans-serif; --fmono:'DM Mono',monospace;
}
body,.gradio-container{background:var(--bg)!important;color:var(--text)!important;font-family:var(--fmono)!important;}
#app-header{background:linear-gradient(135deg,#0f172a 0%,#1e1b4b 55%,#0f172a 100%);border:1px solid #312e81;border-radius:var(--radius);padding:2rem 2.5rem 1.8rem;margin-bottom:1.5rem;position:relative;overflow:hidden;}
#app-header::before{content:'';position:absolute;inset:0;background:radial-gradient(ellipse 65% 65% at 75% 35%,rgba(110,231,183,.09) 0%,transparent 70%);pointer-events:none;}
#app-header h1{font-family:var(--fhead);font-size:2.4rem;font-weight:800;letter-spacing:-.03em;background:linear-gradient(90deg,var(--accent),var(--accent2));-webkit-background-clip:text;-webkit-text-fill-color:transparent;margin:0 0 .35rem;}
#app-header p{color:var(--muted);font-size:.85rem;margin:0;}
#device-badge{display:inline-block;background:var(--surface2);border:1px solid var(--border2);border-radius:99px;padding:.22rem .9rem;font-size:.72rem;color:var(--accent);margin-top:.65rem;}
.panel{background:var(--surface)!important;border:1px solid var(--border)!important;border-radius:var(--radius)!important;padding:1.25rem!important;}
label,.gr-form>label{color:var(--muted)!important;font-size:.76rem!important;letter-spacing:.07em!important;text-transform:uppercase!important;}
textarea,input[type=text],input[type=number]{background:var(--surface2)!important;border:1px solid var(--border2)!important;border-radius:8px!important;color:var(--text)!important;font-family:var(--fmono)!important;font-size:.88rem!important;}
textarea:focus,input:focus{border-color:var(--accent2)!important;box-shadow:0 0 0 2px rgba(56,189,248,.15)!important;outline:none!important;}
input[type=range]{accent-color:var(--accent);}
#generate-btn{background:linear-gradient(135deg,#059669,#0284c7)!important;border:none!important;border-radius:10px!important;color:#fff!important;font-family:var(--fhead)!important;font-weight:700!important;font-size:1rem!important;letter-spacing:.04em!important;padding:.85rem 2rem!important;width:100%!important;cursor:pointer!important;transition:opacity .18s,transform .1s!important;}
#generate-btn:hover{opacity:.87;transform:translateY(-1px);}
#generate-btn:active{transform:translateY(0);}
#stats-box{background:var(--surface2);border:1px solid var(--border);border-radius:8px;padding:.65rem 1rem;font-size:.78rem;color:var(--muted);min-height:2.4rem;margin-top:.4rem;}
audio{width:100%!important;border-radius:8px;accent-color:var(--accent);}
.gr-accordion{background:var(--surface)!important;border:1px solid var(--border)!important;border-radius:var(--radius)!important;}
::-webkit-scrollbar{width:4px;}::-webkit-scrollbar-track{background:var(--bg);}::-webkit-scrollbar-thumb{background:var(--border2);border-radius:99px;}
"""
# ─────────────────────────────────────────────────────────────────────────────
# GRADIO UI
# ─────────────────────────────────────────────────────────────────────────────
with gr.Blocks(css=_CSS, title="πŸŽ™οΈ Neural TTS") as demo:
gr.HTML(f"""
<div id="app-header">
<h1>πŸŽ™οΈ Neural TTS</h1>
<p>Transformers Β· ONNX Β· GGUF LM β†’ ONNX Decoder + Vocoder (humair025/devo)</p>
<span id="device-badge">{_DEVICE_LABEL}</span>
</div>
""")
with gr.Row(equal_height=False):
# ── Left: inputs ───────────────────────────────────────────────────────
with gr.Column(scale=3, elem_classes="panel"):
text_input = gr.Textbox(
label="Text to synthesise",
placeholder="Type or paste any length of text here…",
lines=6, max_lines=24,
)
with gr.Row():
gender_radio = gr.Radio(
choices=["male", "female", "other"], value="female", label="Speaker gender"
)
preset_dropdown = gr.Dropdown(
choices=_voices_for("female"), value=_default_voice("female"),
label="Voice preset (style embedding)",
)
gender_radio.change(fn=_update_preset_dropdown,
inputs=[gender_radio], outputs=[preset_dropdown])
with gr.Row():
voice_tag_input = gr.Textbox(
label="Voice tag (optional)",
placeholder="e.g. kore or puck (blank = none)",
value="",
)
accent_radio = gr.Radio(choices=["us", "uk"], value="us", label="Accent")
mode_radio = gr.Radio(
choices=["text", "phoneme"], value="text", label="Input mode",
info="'text' for normal text Β· 'phoneme' for IPA / phoneme strings",
)
# LM backend
_lm_choices = ["transformers", "onnx"]
_gguf_note = ""
if _LLAMA_CPP_AVAILABLE:
_lm_choices.append("gguf")
else:
_gguf_note = " (gguf unavailable β€” install llama-cpp-python)"
lm_backend_radio = gr.Radio(
choices=_lm_choices, value="transformers", label="LLM backend",
info=f"transformers = safetensors | onnx = model.onnx{_gguf_note}",
)
with gr.Row():
gguf_quant_dropdown = gr.Dropdown(
choices=GGUF_QUANT_FILES, value=LLM_GGUF_FILE,
label="GGUF quant", visible=False,
)
gguf_gpu_layers = gr.Slider(
minimum=-1, maximum=48, value=_DEFAULT_GGUF_GPU_LAYERS, step=1,
label="GGUF GPU layers (-1 = all)", visible=False,
)
lm_backend_radio.change(fn=_update_gguf_visibility,
inputs=[lm_backend_radio],
outputs=[gguf_quant_dropdown, gguf_gpu_layers])
decoder_variant = gr.Radio(
choices=["fp16 (fast, GPU)", "int8 (small, CPU)"],
value="fp16 (fast, GPU)", label="Decoder ONNX variant",
)
with gr.Accordion("βš™οΈ Generation settings", open=False):
with gr.Row():
temperature = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature")
top_p = gr.Slider(0.5, 1.0, value=0.85, step=0.05, label="Top-p")
with gr.Row():
top_k = gr.Slider(10, 200, value=50, step=10, label="Top-k")
repetition_penalty = gr.Slider(1.0, 2.0, value=1.1, step=0.05, label="Repetition penalty")
with gr.Row():
max_new_tokens = gr.Slider(200, 4000, value=1000, step=100, label="Max new tokens")
num_threads = gr.Slider(0, 16, value=0, step=1, label="ORT / llama.cpp threads (0 = auto)")
with gr.Accordion("πŸ”§ Model paths", open=False):
llm_repo_box = gr.Textbox(value=DEFAULT_LLM_REPO_ID, label="LLM repo ID")
llm_subfolder_box = gr.Textbox(value=DEFAULT_LLM_SUBFOLDER, label="LLM checkpoint subfolder")
devo_repo_box = gr.Textbox(value=DEVO_REPO_ID, label="ONNX codec repo (humair025/devo)")
generate_btn = gr.Button("β–Ά Generate Speech", elem_id="generate-btn")
gr.Examples(
examples=[
["The quick brown fox jumps over the lazy dog.", "female", _default_voice("female"), "kore", "us", "text"],
["Welcome to the future of speech synthesis.", "male", _default_voice("male"), "puck", "us", "text"],
["Every voice carries its own unique character.", "female", _default_voice("female"), "", "uk", "text"],
],
inputs=[text_input, gender_radio, preset_dropdown, voice_tag_input, accent_radio, mode_radio],
label="Quick examples",
)
# ── Right: output ──────────────────────────────────────────────────────
with gr.Column(scale=2, elem_classes="panel"):
audio_out = gr.Audio(label="Generated Audio", type="filepath", interactive=False)
stats_md = gr.Markdown(value="*Generation stats will appear here.*", elem_id="stats-box")
gr.HTML("""
<div style="margin-top:1.4rem;padding:.9rem 1.1rem;background:#0f172a;
border:1px solid #1e293b;border-radius:10px;font-size:.75rem;
color:#475569;line-height:1.8;">
<strong style="color:#64748b;font-family:'Syne',sans-serif;">Architecture</strong><br>
LLM generates speech tokens β†’ <b>ONNX Decoder</b> (humair025/devo) converts to mel β†’
<b>HiFT vocoder</b> synthesises audio.<br><br>
<strong style="color:#64748b;font-family:'Syne',sans-serif;">LLM backends</strong><br>
β€’ <b>transformers</b> β€” safetensors from
<code>zuhri025/tts_weights_text_only_soprano / checkpoint-18000</code>.<br>
β€’ <b>onnx</b> β€” FP16 export from <code>humair025/llm_onnx</code>.<br>
β€’ <b>gguf</b> β€” quantised via llama-cpp-python from <code>humair025/llm_gguf</code>.
Q4_K_M β‰ˆ 61 MB. Requires <code>llama-cpp-python</code>.<br><br>
<strong style="color:#64748b;font-family:'Syne',sans-serif;">Voice tags</strong><br>
Type <code>kore</code> or <code>puck</code> (no angle brackets needed).<br><br>
<strong style="color:#64748b;font-family:'Syne',sans-serif;">Notes</strong><br>
β€’ Models cache in memory β€” first run downloads, subsequent runs are instant.<br>
β€’ Voice presets auto-download from <code>humair025/voices</code>.<br>
β€’ GGUF GPU layers: <code>-1</code> = all on GPU Β· <code>0</code> = CPU only.
</div>
""")
generate_btn.click(
fn=generate_speech,
inputs=[
text_input, gender_radio, preset_dropdown,
voice_tag_input, accent_radio, mode_radio,
lm_backend_radio, gguf_quant_dropdown, gguf_gpu_layers,
decoder_variant,
temperature, top_p, top_k, repetition_penalty, max_new_tokens, num_threads,
llm_repo_box, llm_subfolder_box, devo_repo_box,
],
outputs=[audio_out, stats_md],
api_name="generate",
)
if __name__ == "__main__":
demo.launch(
server_name="0.0.0.0",
server_port=int(os.environ.get("PORT", 7860)),
share=True,
)