Streaming voice pipeline: sentence-level TTS, zero temp files

- playback.zig: sentence queue + playback thread — agent reply chunks are
  split on sentence boundaries and each sentence is synthesized+played as it
  completes, so voice starts after the first sentence, not after the reply
- melo_server.py v2: WAV written to /dev/shm, raw s16le PCM + sample rate
  streamed over stdout (WAV n rate header + bytes), file deleted immediately;
  melo stdout chatter rerouted to stderr to keep the protocol clean
- melo.zig: speakRaw returns in-memory PCM; pw-play fed via stdin pipe
  (no on-disk TTS files at all)
- whisper: --language auto
- fix: double-close of pw-play stdin panicked Threaded Io in debug
- fix: quit-path frees for transcript cache, bubble lists, sentence queue,
  reply buffer (debug allocator leak panic)

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
pierreandLetta Code committed 2026-09-01 20:48:16 +03:00
1 parent 2b2235ff86
commit bcd48a3443
5 files changed
+244 -55

No files matched your search

+35 -23
View File
@@ -5,6 +5,7 @@ const std = @import("std");
const Io = std.Io;
const whisper = @import("whisper.zig");
const melo = @import("melo.zig");
const playback = @import("playback.zig");
const qt6 = @import("libqt6zig");
const QApplication = qt6.QApplication;
const QSystemTrayIcon = qt6.QSystemTrayIcon;
@@ -122,6 +123,12 @@ fn onLettaEvent(_: ?*anyopaque, ev: letta.Event) void {
.chunk => |c| {
const dup = allocator.dupe(u8, c) catch return;
pushOverlay(.chunk, dup);
// Stream voice: queue each sentence as it completes.
sentence_buf.appendSlice(allocator, c) catch {};
if (c.len > 0) {
const last = c[c.len - 1];
if (last == '.' or last == '!' or last == '?') flushSentence(false);
}
},
}
}
@@ -186,20 +193,24 @@ fn setState(next: State) void {
}
}
/// Play a WAV via pw-play (blocking; audio length). Fine for v1 — same
/// thread already blocks on whisper/letta. UI-thread decoupling comes later.
fn playWav(path: []const u8) void {
const io = io_ctx orelse return;
var child = std.process.spawn(io, .{
.argv = &.{ "pw-play", path },
.stdin = .ignore,
.stdout = .ignore,
.stderr = .ignore,
}) catch |e| {
std.debug.print("inferon: pw-play spawn failed: {s}\n", .{@errorName(e)});
/// Buffer for sentence-split streaming into the playback pipeline.
var sentence_buf: std.ArrayList(u8) = .empty;
fn flushSentence(final_flush: bool) void {
const a = allocator;
if (sentence_buf.items.len == 0) {
if (final_flush) playback.endReply();
return;
};
_ = child.wait(io) catch {};
}
// Sentences end on . ! ? — hold back a bare trailing terminator-less tail
// unless this is the end of the reply.
if (!final_flush) {
const last = sentence_buf.items[sentence_buf.items.len - 1];
if (last != '.' and last != '!' and last != '?') return;
}
const s = sentence_buf.toOwnedSlice(a) catch return;
playback.push(s);
if (final_flush) playback.endReply();
}
// --------------------------------------------------------------- capture ---
@@ -321,17 +332,10 @@ fn processUtterance() void {
defer allocator.free(response);
std.debug.print("[INFERON]\n{s}\n\n", .{response});
// --- tts + playback ---
// --- tts + playback: sentences already streaming; flush the tail ---
setState(.speaking);
var duration: f64 = 0;
if (melo.speak(allocator, io, response, &duration)) |tts_wav| {
defer allocator.free(tts_wav);
std.debug.print("inferon: tts ready {s} ({d:.2}s synth)\n", .{ tts_wav, duration });
playWav(tts_wav);
cleanupFile(tts_wav);
} else |e| {
std.debug.print("inferon: tts failed: {s}\n", .{@errorName(e)});
}
flushSentence(true);
playback.waitDrained();
// After speaking: back to idle-but-in-conversation. User clicks when
// they want to talk again.
@@ -476,12 +480,20 @@ pub fn main(init: std.process.Init) !void {
};
if (melo.waitReady(allocator, init.io)) {
std.debug.print("inferon: melo ready\n", .{});
playback.start(allocator, init.io);
} else {
std.debug.print("inferon: melo NOT ready\n", .{});
}
_ = QApplication.exec();
// Session-lifetime buffers: free so the debug allocator doesn't squawk.
if (last_user_text) |t| allocator.free(t);
last_user_text = null;
sentence_buf.deinit(allocator);
overlay_mod.Overlay.shutdownMem(); // static, no instance needed
playback.shutdown();
whisper.stop();
melo.stop();
}