Created
September 13, 2026 10:25
-
-
Save 45deg/9dd46b07fed41bb19a99ec1e60db2399 to your computer and use it in GitHub Desktop.
Apple Siliconで動く IrodoriTTS のデモ
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # /// script | |
| # # uv run --script app.py で使う依存関係(PEP 723)。 | |
| # # pyproject.toml の実行時依存関係と揃えてください。 | |
| # requires-python = ">=3.11,<3.14" | |
| # dependencies = [ | |
| # "mlx-audio[tts]==0.5.3", # Irodori推論とTTS用の追加依存関係。 | |
| # "gradio>=6.0,<7", # ローカルの操作画面。 | |
| # "soundfile>=0.13,<0.14", # 音声の検証とWAV保存。 | |
| # ] | |
| # /// | |
| """Run with uv run python app.py (macOS / Apple Silicon).""" | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import logging | |
| import os | |
| from pathlib import Path | |
| import platform | |
| import threading | |
| import time | |
| from uuid import uuid4 | |
| ROOT = Path(__file__).resolve().parent | |
| os.environ.setdefault("HF_HOME", str(ROOT / ".cache" / "huggingface")) | |
| os.environ.setdefault("GRADIO_ANALYTICS_ENABLED", "False") | |
| os.environ.setdefault("GRADIO_TEMP_DIR", str(ROOT / ".cache" / "gradio")) | |
| import gradio as gr | |
| import numpy as np | |
| import soundfile as sf | |
| MODEL_ID = "mlx-community/Irodori-TTS-v4.1-Small-fp16" | |
| OUTPUTS = ROOT / "outputs" | |
| _model = None | |
| _lock = threading.Lock() | |
| def validate(text, caption, reference, duration, scale, seed): | |
| text, caption = text.strip(), caption.strip() | |
| if not text: | |
| raise ValueError("読み上げる文章を入力してください。") | |
| if len(text) > 300: | |
| raise ValueError("文章は300文字以内にしてください。長い文章は分けて生成してください。") | |
| if not caption and not reference: | |
| raise ValueError("声の説明を入力するか、参照音声を選んでください。") | |
| if len(caption) > 500: | |
| raise ValueError("声の説明は500文字以内にしてください。") | |
| if not np.isfinite(duration) or not (duration == 0 or 0.5 <= duration <= 30): | |
| raise ValueError("長さは自動(0)、または0.5〜30秒で指定してください。") | |
| if not np.isfinite(scale) or not 0.5 <= scale <= 1.5: | |
| raise ValueError("長さの倍率は0.5〜1.5で指定してください。") | |
| if not np.isfinite(seed) or int(seed) != seed or not 0 <= seed <= 2**32 - 1: | |
| raise ValueError("シードは0〜4294967295の整数で指定してください。") | |
| if reference: | |
| info = sf.info(reference) | |
| if not 0 < info.duration <= 120: | |
| raise ValueError("参照音声は120秒以内の音声にしてください。") | |
| return text, caption | |
| def synthesize(text, caption, reference, duration, scale, seed, progress=gr.Progress()): | |
| global _model | |
| try: | |
| text, caption = validate(text, caption, reference, duration, scale, seed) | |
| with _lock: | |
| started = time.perf_counter() | |
| if _model is None: | |
| progress(None, desc="モデルを準備中(初回はダウンロードします)") | |
| from mlx_audio.tts.utils import load_model | |
| _model = load_model(MODEL_ID) | |
| progress(None, desc="音声を生成中(完了まで進捗率は表示されません)") | |
| chunks = [] | |
| rate = None | |
| for result in _model.generate( | |
| text=text, caption=caption or None, ref_audio=reference, | |
| seconds=duration or None, duration_scale=scale, | |
| rng_seed=int(seed), max_seconds=30.0, | |
| ): | |
| chunks.append(np.asarray(result.audio, dtype=np.float32).reshape(-1)) | |
| rate = result.sample_rate | |
| if not chunks or not rate: | |
| raise RuntimeError("モデルから音声が返されませんでした。") | |
| audio = np.concatenate(chunks) | |
| if audio.size == 0 or not np.isfinite(audio).all(): | |
| raise RuntimeError("生成音声が空、または無効です。設定を変えて再度お試しください。") | |
| progress(None, desc="WAVを保存中") | |
| OUTPUTS.mkdir(exist_ok=True) | |
| filename = OUTPUTS / f"irodori-{time.strftime('%Y%m%d-%H%M%S')}-{uuid4().hex[:8]}.wav" | |
| sf.write(filename, audio, rate, subtype="PCM_16") | |
| elapsed = time.perf_counter() - started | |
| filename.with_suffix(".json").write_text(json.dumps({ | |
| "model": MODEL_ID, "text": text, "caption": caption, | |
| "reference_used": bool(reference), "seconds": duration or None, | |
| "duration_scale": scale, "seed": int(seed), "sample_rate": rate, | |
| "audio_seconds": len(audio) / rate, "elapsed_seconds": elapsed, | |
| }, ensure_ascii=False, indent=2), encoding="utf-8") | |
| return str(filename), str(filename), f"生成完了 · {len(audio) / rate:.1f}秒の音声 · 処理時間 {elapsed:.1f}秒" | |
| except Exception as exc: | |
| logging.exception("音声生成に失敗しました") | |
| raise gr.Error(f"音声を生成できませんでした: {exc}") from exc | |
| def build_app(): | |
| with gr.Blocks(title="Irodori — 音声をつくる") as demo: | |
| gr.Markdown("# Irodori\n文章から、声をつくる。Apple Siliconで動く日本語音声合成。") | |
| with gr.Row(): | |
| with gr.Column(scale=3): | |
| text = gr.Textbox(label="読み上げる文章", lines=7, max_length=300, | |
| placeholder="ここに日本語の文章を入力してください。", | |
| value="こんにちは。今日は、どんな一日でしたか。あなたの言葉を、自然な声でお届けします。") | |
| caption = gr.Textbox(label="声の説明", lines=2, max_length=500, | |
| value="落ち着いた、やわらかい女性の声。自然な口調で、はっきりと話す。") | |
| reference = gr.Audio(label="参照音声(任意・120秒以内)", sources=["upload"], | |
| type="filepath", format="wav") | |
| gr.Markdown("参照音声を使うと、その声を手がかりに生成します。声の説明と併用できます。") | |
| with gr.Accordion("生成設定", open=False): | |
| duration = gr.Number(label="音声の長さ(秒)・0で自動", value=0, minimum=0, maximum=30) | |
| scale = gr.Slider(0.5, 1.5, value=1, step=0.05, label="自動予測する長さの倍率", | |
| info="自動の場合のみ有効。小さくすると短くなります。") | |
| seed = gr.Number(label="シード(同じ設定で再生成するための数値)", value=0, precision=0, | |
| minimum=0, maximum=2**32 - 1) | |
| generate = gr.Button("音声を生成", variant="primary") | |
| with gr.Column(scale=2): | |
| gr.Markdown("## 生成した音声") | |
| player = gr.Audio(label="プレビュー", type="filepath", interactive=False) | |
| download = gr.File(label="WAVをダウンロード", interactive=False) | |
| status = gr.Textbox(label="生成結果", value="文章と声を設定して、生成してください。", interactive=False) | |
| gr.Markdown("初回はモデルをダウンロードするため、時間がかかります。2回目以降は読み込んだモデルを使います。\n\n短い文で繰り返しが出る場合は、参照音声を使うか、生成設定で長さを調整してください。1回の生成は最大30秒です。") | |
| gr.Markdown(f"モデル: `{MODEL_ID}` \n生成したWAVと設定は、このアプリの `outputs/` に保存します。") | |
| generate.click(synthesize, [text, caption, reference, duration, scale, seed], | |
| [player, download, status], concurrency_limit=1, concurrency_id="inference") | |
| return demo | |
| if __name__ == "__main__": | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--port", type=int, default=7860) | |
| parser.add_argument("--no-browser", action="store_true") | |
| args = parser.parse_args() | |
| if platform.system() != "Darwin" or platform.machine() != "arm64": | |
| parser.error("Apple Silicon搭載のMacで実行してください。") | |
| build_app().queue(max_size=8).launch( | |
| server_name="127.0.0.1", server_port=args.port, share=False, | |
| inbrowser=not args.no_browser, theme=gr.themes.Soft(primary_hue="teal", neutral_hue="slate").set( | |
| block_label_text_color="*primary_800", block_title_text_color="*primary_800", | |
| button_primary_background_fill="*primary_700", | |
| button_primary_background_fill_hover="*primary_800", | |
| input_border_color="*neutral_400", | |
| ), | |
| css=".gradio-container { max-width: 1100px !important; }", | |
| ) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment