Files
inferon/scripts/melo_server.py
T
pierreandLetta Code bcd48a3443 Streaming voice pipeline: sentence-level TTS, zero temp files
- playback.zig: sentence queue + playback thread — agent reply chunks are
  split on sentence boundaries and each sentence is synthesized+played as it
  completes, so voice starts after the first sentence, not after the reply
- melo_server.py v2: WAV written to /dev/shm, raw s16le PCM + sample rate
  streamed over stdout (WAV n rate header + bytes), file deleted immediately;
  melo stdout chatter rerouted to stderr to keep the protocol clean
- melo.zig: speakRaw returns in-memory PCM; pw-play fed via stdin pipe
  (no on-disk TTS files at all)
- whisper: --language auto
- fix: double-close of pw-play stdin panicked Threaded Io in debug
- fix: quit-path frees for transcript cache, bubble lists, sentence queue,
  reply buffer (debug allocator leak panic)

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
2026-09-01 20:48:16 +03:00

112 lines
3.7 KiB
Python

#!/usr/bin/env python3
"""MeloTTS server for inferon.
Long-running subprocess: loads the MeloTTS model once, then reads lines of
text on stdin and writes "<wav-path> <duration-seconds>" per line on stdout.
stderr passes through for logging.
Protocol (v2, raw PCM):
-> one line of text to synthesize
<- header: "WAV <n_bytes> <sample_rate>\n" followed by <n_bytes> of raw s16le PCM
Synthesizes to /dev/shm (RAM-backed), streams bytes, deletes immediately.
Kill with SIGTERM.
"""
import sys
import os
import io
import time
import wave
import signal
import tempfile
import warnings
warnings.filterwarnings("ignore")
MELO_ROOT = os.path.expanduser("~/Projects/MeloTTS")
CKPT = os.path.join(MELO_ROOT, "ckpt/MeloTTS-English-v3/checkpoint.pth")
CONFIG = os.path.join(MELO_ROOT, "ckpt/MeloTTS-English-v3/config.json")
OUT_DIR = "/tmp/inferon"
SPEAKER = "EN-Newest" # v3 English: single speaker (British-accented female)
LANG = "EN"
SPEED = 1.3 # speech rate; 1.0 = natural, higher = faster
sys.path.insert(0, MELO_ROOT)
def synth_bytes(path):
"""Read a WAV, return (raw s16le PCM, sample_rate), delete the file."""
with wave.open(path, "rb") as w:
rate = w.getframerate()
pcm = w.readframes(w.getnframes())
os.unlink(path)
return pcm, rate
def main() -> None:
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0))
signal.signal(signal.SIGINT, lambda *_: sys.exit(0))
os.makedirs(OUT_DIR, exist_ok=True)
# Protocol handle — melo's progress chatter goes to stdout, so we keep
# this reference and then point stdout at stderr.
proto_out = sys.stdout.buffer
# Heavy imports inside main so signal handlers are installed first.
from melo.api import TTS
model = TTS(language=LANG, config_path=CONFIG, ckpt_path=CKPT)
speaker_ids = model.hps.data.spk2id
if SPEAKER not in speaker_ids:
# English v3 fallback: use the first speaker and say so on stderr.
fallback = next(iter(speaker_ids))
print(f"melo_server: speaker {SPEAKER!r} not in {list(speaker_ids)}; using {fallback!r}",
file=sys.stderr, flush=True)
speaker = fallback
else:
speaker = speaker_ids[SPEAKER]
# melo's progress chatter goes to stdout — keep the protocol clean.
sys.stdout = sys.stderr
# Warm up with a full sentence (longer text paths hit NLTK/num2words,
# which is where missing-data errors surface). Fail before READY.
fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm")
os.close(fd)
try:
model.tts_to_file("Warmup sentence number twelve, spoken on the first of January.",
speaker, wpath, speed=SPEED)
_ = synth_bytes(wpath) # exercise the full byte path before READY
finally:
if os.path.exists(wpath):
os.unlink(wpath)
# Ready marker — inferon waits for this before declaring the backend up.
proto_out.write(b"READY\n")
proto_out.flush()
for line in sys.stdin:
text = line.strip()
if not text:
continue
text = text[:2000] # protocol cap — extremely long replies get clipped
try:
fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm")
os.close(fd)
model.tts_to_file(text, speaker, wpath, speed=SPEED)
pcm, rate = synth_bytes(wpath)
proto_out.write(f"WAV {len(pcm)} {rate}\n".encode())
proto_out.write(pcm)
proto_out.flush()
except Exception as e: # noqa: BLE001 - protocol: report and survive
print(f"melo_server: synthesis failed: {e}", file=sys.stderr, flush=True)
print("ERROR", flush=True)
if __name__ == "__main__":
main()