Skip to content

Instantly share code, notes, and snippets.

@45deg
Created September 13, 2026 10:25
Show Gist options
  • Select an option

  • Save 45deg/9dd46b07fed41bb19a99ec1e60db2399 to your computer and use it in GitHub Desktop.

Select an option

Save 45deg/9dd46b07fed41bb19a99ec1e60db2399 to your computer and use it in GitHub Desktop.
Apple Siliconで動く IrodoriTTS のデモ
# /// script
# # uv run --script app.py で使う依存関係(PEP 723)。
# # pyproject.toml の実行時依存関係と揃えてください。
# requires-python = ">=3.11,<3.14"
# dependencies = [
# "mlx-audio[tts]==0.5.3", # Irodori推論とTTS用の追加依存関係。
# "gradio>=6.0,<7", # ローカルの操作画面。
# "soundfile>=0.13,<0.14", # 音声の検証とWAV保存。
# ]
# ///
"""Run with uv run python app.py (macOS / Apple Silicon)."""
from __future__ import annotations
import argparse
import json
import logging
import os
from pathlib import Path
import platform
import threading
import time
from uuid import uuid4
ROOT = Path(__file__).resolve().parent
os.environ.setdefault("HF_HOME", str(ROOT / ".cache" / "huggingface"))
os.environ.setdefault("GRADIO_ANALYTICS_ENABLED", "False")
os.environ.setdefault("GRADIO_TEMP_DIR", str(ROOT / ".cache" / "gradio"))
import gradio as gr
import numpy as np
import soundfile as sf
MODEL_ID = "mlx-community/Irodori-TTS-v4.1-Small-fp16"
OUTPUTS = ROOT / "outputs"
_model = None
_lock = threading.Lock()
def validate(text, caption, reference, duration, scale, seed):
text, caption = text.strip(), caption.strip()
if not text:
raise ValueError("読み上げる文章を入力してください。")
if len(text) > 300:
raise ValueError("文章は300文字以内にしてください。長い文章は分けて生成してください。")
if not caption and not reference:
raise ValueError("声の説明を入力するか、参照音声を選んでください。")
if len(caption) > 500:
raise ValueError("声の説明は500文字以内にしてください。")
if not np.isfinite(duration) or not (duration == 0 or 0.5 <= duration <= 30):
raise ValueError("長さは自動(0)、または0.5〜30秒で指定してください。")
if not np.isfinite(scale) or not 0.5 <= scale <= 1.5:
raise ValueError("長さの倍率は0.5〜1.5で指定してください。")
if not np.isfinite(seed) or int(seed) != seed or not 0 <= seed <= 2**32 - 1:
raise ValueError("シードは0〜4294967295の整数で指定してください。")
if reference:
info = sf.info(reference)
if not 0 < info.duration <= 120:
raise ValueError("参照音声は120秒以内の音声にしてください。")
return text, caption
def synthesize(text, caption, reference, duration, scale, seed, progress=gr.Progress()):
global _model
try:
text, caption = validate(text, caption, reference, duration, scale, seed)
with _lock:
started = time.perf_counter()
if _model is None:
progress(None, desc="モデルを準備中(初回はダウンロードします)")
from mlx_audio.tts.utils import load_model
_model = load_model(MODEL_ID)
progress(None, desc="音声を生成中(完了まで進捗率は表示されません)")
chunks = []
rate = None
for result in _model.generate(
text=text, caption=caption or None, ref_audio=reference,
seconds=duration or None, duration_scale=scale,
rng_seed=int(seed), max_seconds=30.0,
):
chunks.append(np.asarray(result.audio, dtype=np.float32).reshape(-1))
rate = result.sample_rate
if not chunks or not rate:
raise RuntimeError("モデルから音声が返されませんでした。")
audio = np.concatenate(chunks)
if audio.size == 0 or not np.isfinite(audio).all():
raise RuntimeError("生成音声が空、または無効です。設定を変えて再度お試しください。")
progress(None, desc="WAVを保存中")
OUTPUTS.mkdir(exist_ok=True)
filename = OUTPUTS / f"irodori-{time.strftime('%Y%m%d-%H%M%S')}-{uuid4().hex[:8]}.wav"
sf.write(filename, audio, rate, subtype="PCM_16")
elapsed = time.perf_counter() - started
filename.with_suffix(".json").write_text(json.dumps({
"model": MODEL_ID, "text": text, "caption": caption,
"reference_used": bool(reference), "seconds": duration or None,
"duration_scale": scale, "seed": int(seed), "sample_rate": rate,
"audio_seconds": len(audio) / rate, "elapsed_seconds": elapsed,
}, ensure_ascii=False, indent=2), encoding="utf-8")
return str(filename), str(filename), f"生成完了 · {len(audio) / rate:.1f}秒の音声 · 処理時間 {elapsed:.1f}秒"
except Exception as exc:
logging.exception("音声生成に失敗しました")
raise gr.Error(f"音声を生成できませんでした: {exc}") from exc
def build_app():
with gr.Blocks(title="Irodori — 音声をつくる") as demo:
gr.Markdown("# Irodori\n文章から、声をつくる。Apple Siliconで動く日本語音声合成。")
with gr.Row():
with gr.Column(scale=3):
text = gr.Textbox(label="読み上げる文章", lines=7, max_length=300,
placeholder="ここに日本語の文章を入力してください。",
value="こんにちは。今日は、どんな一日でしたか。あなたの言葉を、自然な声でお届けします。")
caption = gr.Textbox(label="声の説明", lines=2, max_length=500,
value="落ち着いた、やわらかい女性の声。自然な口調で、はっきりと話す。")
reference = gr.Audio(label="参照音声(任意・120秒以内)", sources=["upload"],
type="filepath", format="wav")
gr.Markdown("参照音声を使うと、その声を手がかりに生成します。声の説明と併用できます。")
with gr.Accordion("生成設定", open=False):
duration = gr.Number(label="音声の長さ(秒)・0で自動", value=0, minimum=0, maximum=30)
scale = gr.Slider(0.5, 1.5, value=1, step=0.05, label="自動予測する長さの倍率",
info="自動の場合のみ有効。小さくすると短くなります。")
seed = gr.Number(label="シード(同じ設定で再生成するための数値)", value=0, precision=0,
minimum=0, maximum=2**32 - 1)
generate = gr.Button("音声を生成", variant="primary")
with gr.Column(scale=2):
gr.Markdown("## 生成した音声")
player = gr.Audio(label="プレビュー", type="filepath", interactive=False)
download = gr.File(label="WAVをダウンロード", interactive=False)
status = gr.Textbox(label="生成結果", value="文章と声を設定して、生成してください。", interactive=False)
gr.Markdown("初回はモデルをダウンロードするため、時間がかかります。2回目以降は読み込んだモデルを使います。\n\n短い文で繰り返しが出る場合は、参照音声を使うか、生成設定で長さを調整してください。1回の生成は最大30秒です。")
gr.Markdown(f"モデル: `{MODEL_ID}` \n生成したWAVと設定は、このアプリの `outputs/` に保存します。")
generate.click(synthesize, [text, caption, reference, duration, scale, seed],
[player, download, status], concurrency_limit=1, concurrency_id="inference")
return demo
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("--port", type=int, default=7860)
parser.add_argument("--no-browser", action="store_true")
args = parser.parse_args()
if platform.system() != "Darwin" or platform.machine() != "arm64":
parser.error("Apple Silicon搭載のMacで実行してください。")
build_app().queue(max_size=8).launch(
server_name="127.0.0.1", server_port=args.port, share=False,
inbrowser=not args.no_browser, theme=gr.themes.Soft(primary_hue="teal", neutral_hue="slate").set(
block_label_text_color="*primary_800", block_title_text_color="*primary_800",
button_primary_background_fill="*primary_700",
button_primary_background_fill_hover="*primary_800",
input_border_color="*neutral_400",
),
css=".gradio-container { max-width: 1100px !important; }",
)
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment