#!/usr/bin/env python3 """MeloTTS server for inferon. Long-running subprocess: loads the MeloTTS model once, then reads lines of text on stdin and writes " " per line on stdout. stderr passes through for logging. Protocol (v2, raw PCM): -> one line of text to synthesize <- header: "WAV \n" followed by of raw s16le PCM Synthesizes to /dev/shm (RAM-backed), streams bytes, deletes immediately. Kill with SIGTERM. """ import sys import os import io import time import wave import signal import tempfile import warnings warnings.filterwarnings("ignore") MELO_ROOT = os.path.expanduser("~/Projects/MeloTTS") CKPT = os.path.join(MELO_ROOT, "ckpt/MeloTTS-English-v3/checkpoint.pth") CONFIG = os.path.join(MELO_ROOT, "ckpt/MeloTTS-English-v3/config.json") OUT_DIR = "/tmp/inferon" SPEAKER = "EN-Newest" # v3 English: single speaker (British-accented female) LANG = "EN" SPEED = 1.3 # speech rate; 1.0 = natural, higher = faster sys.path.insert(0, MELO_ROOT) def synth_bytes(path): """Read a WAV, return (raw s16le PCM, sample_rate), delete the file.""" with wave.open(path, "rb") as w: rate = w.getframerate() pcm = w.readframes(w.getnframes()) os.unlink(path) return pcm, rate def main() -> None: signal.signal(signal.SIGTERM, lambda *_: sys.exit(0)) signal.signal(signal.SIGINT, lambda *_: sys.exit(0)) os.makedirs(OUT_DIR, exist_ok=True) # Protocol handle — melo's progress chatter goes to stdout, so we keep # this reference and then point stdout at stderr. proto_out = sys.stdout.buffer # Heavy imports inside main so signal handlers are installed first. from melo.api import TTS model = TTS(language=LANG, config_path=CONFIG, ckpt_path=CKPT) speaker_ids = model.hps.data.spk2id if SPEAKER not in speaker_ids: # English v3 fallback: use the first speaker and say so on stderr. fallback = next(iter(speaker_ids)) print(f"melo_server: speaker {SPEAKER!r} not in {list(speaker_ids)}; using {fallback!r}", file=sys.stderr, flush=True) speaker = fallback else: speaker = speaker_ids[SPEAKER] # melo's progress chatter goes to stdout — keep the protocol clean. sys.stdout = sys.stderr # Warm up with a full sentence (longer text paths hit NLTK/num2words, # which is where missing-data errors surface). Fail before READY. fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm") os.close(fd) try: model.tts_to_file("Warmup sentence number twelve, spoken on the first of January.", speaker, wpath, speed=SPEED) _ = synth_bytes(wpath) # exercise the full byte path before READY finally: if os.path.exists(wpath): os.unlink(wpath) # Ready marker — inferon waits for this before declaring the backend up. proto_out.write(b"READY\n") proto_out.flush() for line in sys.stdin: text = line.strip() if not text: continue text = text[:2000] # protocol cap — extremely long replies get clipped try: fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm") os.close(fd) model.tts_to_file(text, speaker, wpath, speed=SPEED) pcm, rate = synth_bytes(wpath) proto_out.write(f"WAV {len(pcm)} {rate}\n".encode()) proto_out.write(pcm) proto_out.flush() except Exception as e: # noqa: BLE001 - protocol: report and survive print(f"melo_server: synthesis failed: {e}", file=sys.stderr, flush=True) print("ERROR", flush=True) if __name__ == "__main__": main()