Download app.py from humair025/llm67: direct link, hf CLI and curl.
- Browser
- Download file 22.1 kB
-
https://huggingface.co/spaces/humair025/llm67/resolve/main/app.py
- Command line
-
hf download hf://spaces/humair025/llm67/app.py
-
curl -L -o app.py https://huggingface.co/spaces/humair025/llm67/resolve/main/app.py
22.1 kB
| """ | |
| app.py β Gradio frontend for Neural TTS | |
| Transformers / ONNX / GGUF LLM β ONNX Decoder + Vocoder (humair025/devo) | |
| LLM: zuhri025/tts_weights_text_only_soprano | |
| Supports CUDA, MPS, CPU. | |
| """ | |
| import os | |
| os.environ["OMP_NUM_THREADS"] = "2" | |
| os.environ["MKL_NUM_THREADS"] = "2" | |
| os.environ["OPENBLAS_NUM_THREADS"] = "2" | |
| os.environ["VECLIB_MAXIMUM_THREADS"] = "2" | |
| os.environ["NUMEXPR_NUM_THREADS"] = "2" | |
| import tempfile | |
| import numpy as np | |
| import soundfile as sf | |
| import torch | |
| torch.set_num_threads(2) | |
| torch.set_num_interop_threads(2) | |
| import gradio as gr | |
| from tts_engine import ( | |
| text_to_audio, | |
| load_style_embedding, | |
| DEFAULT_LLM_REPO_ID, | |
| DEFAULT_LLM_SUBFOLDER, | |
| DEVO_REPO_ID, | |
| DEFAULT_DECODER_FILE, | |
| DEFAULT_VOCODER_FILE, | |
| LLM_ONNX_REPO_ID, | |
| LLM_ONNX_FILE, | |
| LLM_GGUF_REPO_ID, | |
| LLM_GGUF_FILE, | |
| GGUF_QUANT_FILES, | |
| SAMPLE_RATE, | |
| _LLAMA_CPP_AVAILABLE, | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # DEVICE | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| if torch.cuda.is_available(): | |
| _DEVICE_LABEL = f"π’ GPU β {torch.cuda.get_device_name(0)}" | |
| _USE_AUTOCAST = True | |
| _PIN_MEMORY = True | |
| _DEFAULT_GGUF_GPU_LAYERS = -1 | |
| elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available(): | |
| _DEVICE_LABEL = "π‘ MPS β Apple Silicon" | |
| _USE_AUTOCAST = False | |
| _PIN_MEMORY = False | |
| _DEFAULT_GGUF_GPU_LAYERS = 0 | |
| else: | |
| _DEVICE_LABEL = "π΅ CPU" | |
| _USE_AUTOCAST = False | |
| _PIN_MEMORY = False | |
| _DEFAULT_GGUF_GPU_LAYERS = 0 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # VOICE PRESETS | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| VOICES_HF_REPO = "humair025/voices" | |
| VOICES_GENDERS = ["male", "female"] | |
| _BASE_DIR = os.path.dirname(os.path.abspath(__file__)) | |
| VOICES_ROOT = os.path.join(_BASE_DIR, "voices") | |
| def _hf_list_voice_files(gender: str) -> list: | |
| try: | |
| from huggingface_hub import list_repo_files | |
| supported = (".pt", ".pth", ".npy") | |
| return sorted( | |
| os.path.basename(f) | |
| for f in list_repo_files(VOICES_HF_REPO) | |
| if f.startswith(f"{gender}/") and os.path.splitext(f)[1].lower() in supported | |
| ) | |
| except Exception as e: | |
| print(f"[voices] Could not list HF files for '{gender}': {e}") | |
| return [] | |
| def _ensure_voice_file(gender: str, filename: str) -> str: | |
| local_path = os.path.join(VOICES_ROOT, gender, filename) | |
| if os.path.isfile(local_path): | |
| return local_path | |
| os.makedirs(os.path.join(VOICES_ROOT, gender), exist_ok=True) | |
| try: | |
| from huggingface_hub import hf_hub_download | |
| return hf_hub_download(repo_id=VOICES_HF_REPO, | |
| filename=f"{gender}/{filename}", | |
| local_dir=VOICES_ROOT) | |
| except Exception as e: | |
| print(f"[voices] FAILED to download {gender}/{filename}: {e}") | |
| return local_path | |
| def _load_voices_for_gender(gender: str) -> dict: | |
| voices = {} | |
| supported = (".pt", ".pth", ".npy") | |
| local_dir = os.path.join(VOICES_ROOT, gender) | |
| local_files = set() | |
| if os.path.isdir(local_dir): | |
| local_files = {f for f in os.listdir(local_dir) | |
| if os.path.splitext(f)[1].lower() in supported} | |
| all_files = sorted(local_files | set(_hf_list_voice_files(gender))) | |
| if not all_files: | |
| print(f"[voices] β οΈ No voice files for '{gender}'.") | |
| return voices | |
| for fname in all_files: | |
| path = _ensure_voice_file(gender, fname) | |
| if not os.path.isfile(path): | |
| continue | |
| try: | |
| emb = load_style_embedding(path) | |
| if emb is not None: | |
| voices[os.path.splitext(fname)[0]] = emb | |
| print(f"[voices] β {fname} shape={emb.shape}") | |
| except Exception as e: | |
| print(f"[voices] SKIP {fname}: {e}") | |
| return voices | |
| print("[voices] Loading voice presetsβ¦") | |
| PRESET_EMBEDDINGS: dict = {g: _load_voices_for_gender(g) for g in VOICES_GENDERS} | |
| print("[voices] Ready β " + " ".join( | |
| f"{g}: {len(v)} voice(s)" for g, v in PRESET_EMBEDDINGS.items())) | |
| def _voices_for(gender: str) -> list: | |
| return sorted(PRESET_EMBEDDINGS.get(gender, {}).keys()) | |
| def _default_voice(gender: str): | |
| c = _voices_for(gender) | |
| return c[0] if c else None | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # INFERENCE | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def generate_speech( | |
| text, gender, preset_choice, | |
| voice_tag_input, accent, mode, | |
| lm_backend_type, gguf_quant, gguf_n_gpu_layers, | |
| decoder_variant, | |
| temperature, top_p, top_k, repetition_penalty, max_new_tokens, num_threads, | |
| llm_repo, llm_subfolder, devo_repo, | |
| progress=gr.Progress(track_tqdm=False), | |
| ): | |
| if not text.strip(): | |
| raise gr.Error("Please enter some text to synthesise.") | |
| # Style embedding | |
| style_np = PRESET_EMBEDDINGS.get(gender, {}).get(preset_choice) | |
| if style_np is None and PRESET_EMBEDDINGS.get(gender): | |
| style_np = next(iter(PRESET_EMBEDDINGS[gender].values())) | |
| # Normalise voice tag | |
| vt = voice_tag_input.strip() if voice_tag_input else None | |
| if vt and not vt.startswith("<|"): | |
| vt = f"<|{vt}|>" | |
| if not vt: | |
| vt = None | |
| # Decoder ONNX variant | |
| decoder_file = { | |
| "fp16 (fast, GPU)": "decoder_fp16.onnx", | |
| "int8 (small, CPU)": "decoder_int8.onnx", | |
| }.get(decoder_variant, DEFAULT_DECODER_FILE) | |
| _be_label = {"transformers": "Transformers", "onnx": "ONNX LLM", | |
| "gguf": f"GGUF ({gguf_quant})"}.get(lm_backend_type, lm_backend_type) | |
| progress(0.1, desc=f"Preparing {_be_label} backendβ¦") | |
| with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f: | |
| out_path = f.name | |
| try: | |
| progress(0.2, desc="Generating speechβ¦") | |
| wav_np, sr, stats = text_to_audio( | |
| text = text, | |
| gender = gender, | |
| voice_tag = vt, | |
| accent = accent, | |
| mode = mode, | |
| style_path = style_np, | |
| lm_backend_type = lm_backend_type, | |
| llm_path = llm_repo, | |
| llm_subfolder = llm_subfolder, | |
| llm_onnx_repo = LLM_ONNX_REPO_ID, | |
| llm_onnx_file = LLM_ONNX_FILE, | |
| gguf_repo = LLM_GGUF_REPO_ID, | |
| gguf_file = gguf_quant, | |
| gguf_n_gpu_layers = int(gguf_n_gpu_layers), | |
| gguf_n_ctx = 2048, | |
| decoder_path = decoder_file, | |
| vocoder_path = DEFAULT_VOCODER_FILE, | |
| devo_repo_id = devo_repo, | |
| device = "auto", | |
| num_threads = int(num_threads), | |
| save_path = out_path, | |
| play = False, | |
| verbose = True, | |
| stream_generation = False, | |
| use_autocast = _USE_AUTOCAST, | |
| pin_memory = _PIN_MEMORY, | |
| temperature = temperature, | |
| top_p = top_p, | |
| top_k = int(top_k), | |
| repetition_penalty= repetition_penalty, | |
| max_new_tokens = int(max_new_tokens), | |
| ) | |
| progress(1.0, desc="Done!") | |
| except Exception as e: | |
| raise gr.Error(f"Generation failed: {e}") | |
| rtf = stats.get("rtf_total", 0) | |
| speed = stats.get("speed_x", 0) | |
| dur = stats.get("audio_duration_s", 0) | |
| t_tot = stats.get("t_total_s", 0) | |
| t_lm = stats.get("t_lm_s", 0) | |
| t_dec = stats.get("t_decode_s", 0) | |
| info = ( | |
| f"**Preset:** {preset_choice or 'β'} ({gender}) | " | |
| f"**Voice tag:** {vt or 'β'} | " | |
| f"**Accent:** {accent} | " | |
| f"**LM:** {stats.get('backend', '?')} | " | |
| f"**Duration:** {dur:.2f}s | " | |
| f"**Gen time:** {t_tot:.2f}s (LM {t_lm:.2f}s + decode {t_dec:.2f}s) | " | |
| f"**Speed:** {speed:.2f}Γ RT | " | |
| f"**RTF:** {rtf:.4f}" | |
| ) | |
| return out_path, info | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # UI HELPERS | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _update_preset_dropdown(gender): | |
| choices = _voices_for(gender) | |
| return gr.update(choices=choices, value=choices[0] if choices else None) | |
| def _update_gguf_visibility(lm_backend_type): | |
| vis = lm_backend_type == "gguf" | |
| return gr.update(visible=vis), gr.update(visible=vis) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # CSS | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| _CSS = """ | |
| @import url('https://fonts.googleapis.com/css2?family=Syne:wght@400;600;800&family=DM+Mono:ital,wght@0,400;0,500;1,400&display=swap'); | |
| :root { | |
| --bg:#0b0d12; --surface:#13161f; --surface2:#1b1f2e; | |
| --border:#1e293b; --border2:#334155; | |
| --accent:#6ee7b7; --accent2:#38bdf8; | |
| --text:#e2e8f0; --muted:#64748b; | |
| --radius:12px; --fhead:'Syne',sans-serif; --fmono:'DM Mono',monospace; | |
| } | |
| body,.gradio-container{background:var(--bg)!important;color:var(--text)!important;font-family:var(--fmono)!important;} | |
| #app-header{background:linear-gradient(135deg,#0f172a 0%,#1e1b4b 55%,#0f172a 100%);border:1px solid #312e81;border-radius:var(--radius);padding:2rem 2.5rem 1.8rem;margin-bottom:1.5rem;position:relative;overflow:hidden;} | |
| #app-header::before{content:'';position:absolute;inset:0;background:radial-gradient(ellipse 65% 65% at 75% 35%,rgba(110,231,183,.09) 0%,transparent 70%);pointer-events:none;} | |
| #app-header h1{font-family:var(--fhead);font-size:2.4rem;font-weight:800;letter-spacing:-.03em;background:linear-gradient(90deg,var(--accent),var(--accent2));-webkit-background-clip:text;-webkit-text-fill-color:transparent;margin:0 0 .35rem;} | |
| #app-header p{color:var(--muted);font-size:.85rem;margin:0;} | |
| #device-badge{display:inline-block;background:var(--surface2);border:1px solid var(--border2);border-radius:99px;padding:.22rem .9rem;font-size:.72rem;color:var(--accent);margin-top:.65rem;} | |
| .panel{background:var(--surface)!important;border:1px solid var(--border)!important;border-radius:var(--radius)!important;padding:1.25rem!important;} | |
| label,.gr-form>label{color:var(--muted)!important;font-size:.76rem!important;letter-spacing:.07em!important;text-transform:uppercase!important;} | |
| textarea,input[type=text],input[type=number]{background:var(--surface2)!important;border:1px solid var(--border2)!important;border-radius:8px!important;color:var(--text)!important;font-family:var(--fmono)!important;font-size:.88rem!important;} | |
| textarea:focus,input:focus{border-color:var(--accent2)!important;box-shadow:0 0 0 2px rgba(56,189,248,.15)!important;outline:none!important;} | |
| input[type=range]{accent-color:var(--accent);} | |
| #generate-btn{background:linear-gradient(135deg,#059669,#0284c7)!important;border:none!important;border-radius:10px!important;color:#fff!important;font-family:var(--fhead)!important;font-weight:700!important;font-size:1rem!important;letter-spacing:.04em!important;padding:.85rem 2rem!important;width:100%!important;cursor:pointer!important;transition:opacity .18s,transform .1s!important;} | |
| #generate-btn:hover{opacity:.87;transform:translateY(-1px);} | |
| #generate-btn:active{transform:translateY(0);} | |
| #stats-box{background:var(--surface2);border:1px solid var(--border);border-radius:8px;padding:.65rem 1rem;font-size:.78rem;color:var(--muted);min-height:2.4rem;margin-top:.4rem;} | |
| audio{width:100%!important;border-radius:8px;accent-color:var(--accent);} | |
| .gr-accordion{background:var(--surface)!important;border:1px solid var(--border)!important;border-radius:var(--radius)!important;} | |
| ::-webkit-scrollbar{width:4px;}::-webkit-scrollbar-track{background:var(--bg);}::-webkit-scrollbar-thumb{background:var(--border2);border-radius:99px;} | |
| """ | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # GRADIO UI | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| with gr.Blocks(css=_CSS, title="ποΈ Neural TTS") as demo: | |
| gr.HTML(f""" | |
| <div id="app-header"> | |
| <h1>ποΈ Neural TTS</h1> | |
| <p>Transformers Β· ONNX Β· GGUF LM β ONNX Decoder + Vocoder (humair025/devo)</p> | |
| <span id="device-badge">{_DEVICE_LABEL}</span> | |
| </div> | |
| """) | |
| with gr.Row(equal_height=False): | |
| # ββ Left: inputs βββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| with gr.Column(scale=3, elem_classes="panel"): | |
| text_input = gr.Textbox( | |
| label="Text to synthesise", | |
| placeholder="Type or paste any length of text hereβ¦", | |
| lines=6, max_lines=24, | |
| ) | |
| with gr.Row(): | |
| gender_radio = gr.Radio( | |
| choices=["male", "female", "other"], value="female", label="Speaker gender" | |
| ) | |
| preset_dropdown = gr.Dropdown( | |
| choices=_voices_for("female"), value=_default_voice("female"), | |
| label="Voice preset (style embedding)", | |
| ) | |
| gender_radio.change(fn=_update_preset_dropdown, | |
| inputs=[gender_radio], outputs=[preset_dropdown]) | |
| with gr.Row(): | |
| voice_tag_input = gr.Textbox( | |
| label="Voice tag (optional)", | |
| placeholder="e.g. kore or puck (blank = none)", | |
| value="", | |
| ) | |
| accent_radio = gr.Radio(choices=["us", "uk"], value="us", label="Accent") | |
| mode_radio = gr.Radio( | |
| choices=["text", "phoneme"], value="text", label="Input mode", | |
| info="'text' for normal text Β· 'phoneme' for IPA / phoneme strings", | |
| ) | |
| # LM backend | |
| _lm_choices = ["transformers", "onnx"] | |
| _gguf_note = "" | |
| if _LLAMA_CPP_AVAILABLE: | |
| _lm_choices.append("gguf") | |
| else: | |
| _gguf_note = " (gguf unavailable β install llama-cpp-python)" | |
| lm_backend_radio = gr.Radio( | |
| choices=_lm_choices, value="transformers", label="LLM backend", | |
| info=f"transformers = safetensors | onnx = model.onnx{_gguf_note}", | |
| ) | |
| with gr.Row(): | |
| gguf_quant_dropdown = gr.Dropdown( | |
| choices=GGUF_QUANT_FILES, value=LLM_GGUF_FILE, | |
| label="GGUF quant", visible=False, | |
| ) | |
| gguf_gpu_layers = gr.Slider( | |
| minimum=-1, maximum=48, value=_DEFAULT_GGUF_GPU_LAYERS, step=1, | |
| label="GGUF GPU layers (-1 = all)", visible=False, | |
| ) | |
| lm_backend_radio.change(fn=_update_gguf_visibility, | |
| inputs=[lm_backend_radio], | |
| outputs=[gguf_quant_dropdown, gguf_gpu_layers]) | |
| decoder_variant = gr.Radio( | |
| choices=["fp16 (fast, GPU)", "int8 (small, CPU)"], | |
| value="fp16 (fast, GPU)", label="Decoder ONNX variant", | |
| ) | |
| with gr.Accordion("βοΈ Generation settings", open=False): | |
| with gr.Row(): | |
| temperature = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature") | |
| top_p = gr.Slider(0.5, 1.0, value=0.85, step=0.05, label="Top-p") | |
| with gr.Row(): | |
| top_k = gr.Slider(10, 200, value=50, step=10, label="Top-k") | |
| repetition_penalty = gr.Slider(1.0, 2.0, value=1.1, step=0.05, label="Repetition penalty") | |
| with gr.Row(): | |
| max_new_tokens = gr.Slider(200, 4000, value=1000, step=100, label="Max new tokens") | |
| num_threads = gr.Slider(0, 16, value=0, step=1, label="ORT / llama.cpp threads (0 = auto)") | |
| with gr.Accordion("π§ Model paths", open=False): | |
| llm_repo_box = gr.Textbox(value=DEFAULT_LLM_REPO_ID, label="LLM repo ID") | |
| llm_subfolder_box = gr.Textbox(value=DEFAULT_LLM_SUBFOLDER, label="LLM checkpoint subfolder") | |
| devo_repo_box = gr.Textbox(value=DEVO_REPO_ID, label="ONNX codec repo (humair025/devo)") | |
| generate_btn = gr.Button("βΆ Generate Speech", elem_id="generate-btn") | |
| gr.Examples( | |
| examples=[ | |
| ["The quick brown fox jumps over the lazy dog.", "female", _default_voice("female"), "kore", "us", "text"], | |
| ["Welcome to the future of speech synthesis.", "male", _default_voice("male"), "puck", "us", "text"], | |
| ["Every voice carries its own unique character.", "female", _default_voice("female"), "", "uk", "text"], | |
| ], | |
| inputs=[text_input, gender_radio, preset_dropdown, voice_tag_input, accent_radio, mode_radio], | |
| label="Quick examples", | |
| ) | |
| # ββ Right: output ββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| with gr.Column(scale=2, elem_classes="panel"): | |
| audio_out = gr.Audio(label="Generated Audio", type="filepath", interactive=False) | |
| stats_md = gr.Markdown(value="*Generation stats will appear here.*", elem_id="stats-box") | |
| gr.HTML(""" | |
| <div style="margin-top:1.4rem;padding:.9rem 1.1rem;background:#0f172a; | |
| border:1px solid #1e293b;border-radius:10px;font-size:.75rem; | |
| color:#475569;line-height:1.8;"> | |
| <strong style="color:#64748b;font-family:'Syne',sans-serif;">Architecture</strong><br> | |
| LLM generates speech tokens β <b>ONNX Decoder</b> (humair025/devo) converts to mel β | |
| <b>HiFT vocoder</b> synthesises audio.<br><br> | |
| <strong style="color:#64748b;font-family:'Syne',sans-serif;">LLM backends</strong><br> | |
| β’ <b>transformers</b> β safetensors from | |
| <code>zuhri025/tts_weights_text_only_soprano / checkpoint-18000</code>.<br> | |
| β’ <b>onnx</b> β FP16 export from <code>humair025/llm_onnx</code>.<br> | |
| β’ <b>gguf</b> β quantised via llama-cpp-python from <code>humair025/llm_gguf</code>. | |
| Q4_K_M β 61 MB. Requires <code>llama-cpp-python</code>.<br><br> | |
| <strong style="color:#64748b;font-family:'Syne',sans-serif;">Voice tags</strong><br> | |
| Type <code>kore</code> or <code>puck</code> (no angle brackets needed).<br><br> | |
| <strong style="color:#64748b;font-family:'Syne',sans-serif;">Notes</strong><br> | |
| β’ Models cache in memory β first run downloads, subsequent runs are instant.<br> | |
| β’ Voice presets auto-download from <code>humair025/voices</code>.<br> | |
| β’ GGUF GPU layers: <code>-1</code> = all on GPU Β· <code>0</code> = CPU only. | |
| </div> | |
| """) | |
| generate_btn.click( | |
| fn=generate_speech, | |
| inputs=[ | |
| text_input, gender_radio, preset_dropdown, | |
| voice_tag_input, accent_radio, mode_radio, | |
| lm_backend_radio, gguf_quant_dropdown, gguf_gpu_layers, | |
| decoder_variant, | |
| temperature, top_p, top_k, repetition_penalty, max_new_tokens, num_threads, | |
| llm_repo_box, llm_subfolder_box, devo_repo_box, | |
| ], | |
| outputs=[audio_out, stats_md], | |
| api_name="generate", | |
| ) | |
| if __name__ == "__main__": | |
| demo.launch( | |
| server_name="0.0.0.0", | |
| server_port=int(os.environ.get("PORT", 7860)), | |
| share=True, | |
| ) |