Streaming voice pipeline: sentence-level TTS, zero temp files
- playback.zig: sentence queue + playback thread — agent reply chunks are split on sentence boundaries and each sentence is synthesized+played as it completes, so voice starts after the first sentence, not after the reply - melo_server.py v2: WAV written to /dev/shm, raw s16le PCM + sample rate streamed over stdout (WAV n rate header + bytes), file deleted immediately; melo stdout chatter rerouted to stderr to keep the protocol clean - melo.zig: speakRaw returns in-memory PCM; pw-play fed via stdin pipe (no on-disk TTS files at all) - whisper: --language auto - fix: double-close of pw-play stdin panicked Threaded Io in debug - fix: quit-path frees for transcript cache, bubble lists, sentence queue, reply buffer (debug allocator leak panic) 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
1 parent
2b2235ff86
commit
bcd48a3443
5 files changed
+244
-55
No files matched your search
+41
-13
@@ -5,17 +5,21 @@ Long-running subprocess: loads the MeloTTS model once, then reads lines of
|
||||
text on stdin and writes "<wav-path> <duration-seconds>" per line on stdout.
|
||||
stderr passes through for logging.
|
||||
|
||||
Protocol:
|
||||
Protocol (v2, raw PCM):
|
||||
-> one line of text to synthesize
|
||||
<- e.g. "/tmp/inferon/tts-1723.wav 3.42"
|
||||
<- header: "WAV <n_bytes> <sample_rate>\n" followed by <n_bytes> of raw s16le PCM
|
||||
|
||||
Kill with SIGTERM — no cleanup needed, WAVs live in /tmp.
|
||||
Synthesizes to /dev/shm (RAM-backed), streams bytes, deletes immediately.
|
||||
Kill with SIGTERM.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import io
|
||||
import time
|
||||
import wave
|
||||
import signal
|
||||
import tempfile
|
||||
import warnings
|
||||
|
||||
warnings.filterwarnings("ignore")
|
||||
@@ -31,12 +35,25 @@ SPEED = 1.3 # speech rate; 1.0 = natural, higher = faster
|
||||
sys.path.insert(0, MELO_ROOT)
|
||||
|
||||
|
||||
def synth_bytes(path):
|
||||
"""Read a WAV, return (raw s16le PCM, sample_rate), delete the file."""
|
||||
with wave.open(path, "rb") as w:
|
||||
rate = w.getframerate()
|
||||
pcm = w.readframes(w.getnframes())
|
||||
os.unlink(path)
|
||||
return pcm, rate
|
||||
|
||||
|
||||
def main() -> None:
|
||||
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0))
|
||||
signal.signal(signal.SIGINT, lambda *_: sys.exit(0))
|
||||
|
||||
os.makedirs(OUT_DIR, exist_ok=True)
|
||||
|
||||
# Protocol handle — melo's progress chatter goes to stdout, so we keep
|
||||
# this reference and then point stdout at stderr.
|
||||
proto_out = sys.stdout.buffer
|
||||
|
||||
# Heavy imports inside main so signal handlers are installed first.
|
||||
from melo.api import TTS
|
||||
|
||||
@@ -52,15 +69,24 @@ def main() -> None:
|
||||
else:
|
||||
speaker = speaker_ids[SPEAKER]
|
||||
|
||||
# melo's progress chatter goes to stdout — keep the protocol clean.
|
||||
sys.stdout = sys.stderr
|
||||
|
||||
# Warm up with a full sentence (longer text paths hit NLTK/num2words,
|
||||
# which is where missing-data errors surface). Fail before READY.
|
||||
warmup = "/tmp/inferon/tts-warmup.wav"
|
||||
model.tts_to_file("Warmup sentence number twelve, spoken on the first of January.",
|
||||
speaker, warmup, speed=SPEED)
|
||||
os.unlink(warmup)
|
||||
fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm")
|
||||
os.close(fd)
|
||||
try:
|
||||
model.tts_to_file("Warmup sentence number twelve, spoken on the first of January.",
|
||||
speaker, wpath, speed=SPEED)
|
||||
_ = synth_bytes(wpath) # exercise the full byte path before READY
|
||||
finally:
|
||||
if os.path.exists(wpath):
|
||||
os.unlink(wpath)
|
||||
|
||||
# Ready marker — inferon waits for this before declaring the backend up.
|
||||
print("READY", flush=True)
|
||||
proto_out.write(b"READY\n")
|
||||
proto_out.flush()
|
||||
|
||||
for line in sys.stdin:
|
||||
text = line.strip()
|
||||
@@ -68,12 +94,14 @@ def main() -> None:
|
||||
continue
|
||||
text = text[:2000] # protocol cap — extremely long replies get clipped
|
||||
|
||||
wav_path = os.path.join(OUT_DIR, f"tts-{time.time_ns()}.wav")
|
||||
try:
|
||||
start = time.monotonic()
|
||||
model.tts_to_file(text, speaker, wav_path, speed=SPEED)
|
||||
dur = time.monotonic() - start
|
||||
print(f"{wav_path} {dur:.2f}", flush=True)
|
||||
fd, wpath = tempfile.mkstemp(suffix=".wav", dir="/dev/shm")
|
||||
os.close(fd)
|
||||
model.tts_to_file(text, speaker, wpath, speed=SPEED)
|
||||
pcm, rate = synth_bytes(wpath)
|
||||
proto_out.write(f"WAV {len(pcm)} {rate}\n".encode())
|
||||
proto_out.write(pcm)
|
||||
proto_out.flush()
|
||||
except Exception as e: # noqa: BLE001 - protocol: report and survive
|
||||
print(f"melo_server: synthesis failed: {e}", file=sys.stderr, flush=True)
|
||||
print("ERROR", flush=True)
|
||||
|
||||
Reference in new issue
Block a user