Streaming voice pipeline: sentence-level TTS, zero temp files
- playback.zig: sentence queue + playback thread — agent reply chunks are split on sentence boundaries and each sentence is synthesized+played as it completes, so voice starts after the first sentence, not after the reply - melo_server.py v2: WAV written to /dev/shm, raw s16le PCM + sample rate streamed over stdout (WAV n rate header + bytes), file deleted immediately; melo stdout chatter rerouted to stderr to keep the protocol clean - melo.zig: speakRaw returns in-memory PCM; pw-play fed via stdin pipe (no on-disk TTS files at all) - whisper: --language auto - fix: double-close of pw-play stdin panicked Threaded Io in debug - fix: quit-path frees for transcript cache, bubble lists, sentence queue, reply buffer (debug allocator leak panic) 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
1 parent
2b2235ff86
commit
bcd48a3443
5 files changed
+244
-55
No files matched your search
+35
-23
@@ -5,6 +5,7 @@ const std = @import("std");
|
||||
const Io = std.Io;
|
||||
const whisper = @import("whisper.zig");
|
||||
const melo = @import("melo.zig");
|
||||
const playback = @import("playback.zig");
|
||||
const qt6 = @import("libqt6zig");
|
||||
const QApplication = qt6.QApplication;
|
||||
const QSystemTrayIcon = qt6.QSystemTrayIcon;
|
||||
@@ -122,6 +123,12 @@ fn onLettaEvent(_: ?*anyopaque, ev: letta.Event) void {
|
||||
.chunk => |c| {
|
||||
const dup = allocator.dupe(u8, c) catch return;
|
||||
pushOverlay(.chunk, dup);
|
||||
// Stream voice: queue each sentence as it completes.
|
||||
sentence_buf.appendSlice(allocator, c) catch {};
|
||||
if (c.len > 0) {
|
||||
const last = c[c.len - 1];
|
||||
if (last == '.' or last == '!' or last == '?') flushSentence(false);
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -186,20 +193,24 @@ fn setState(next: State) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Play a WAV via pw-play (blocking; audio length). Fine for v1 — same
|
||||
/// thread already blocks on whisper/letta. UI-thread decoupling comes later.
|
||||
fn playWav(path: []const u8) void {
|
||||
const io = io_ctx orelse return;
|
||||
var child = std.process.spawn(io, .{
|
||||
.argv = &.{ "pw-play", path },
|
||||
.stdin = .ignore,
|
||||
.stdout = .ignore,
|
||||
.stderr = .ignore,
|
||||
}) catch |e| {
|
||||
std.debug.print("inferon: pw-play spawn failed: {s}\n", .{@errorName(e)});
|
||||
/// Buffer for sentence-split streaming into the playback pipeline.
|
||||
var sentence_buf: std.ArrayList(u8) = .empty;
|
||||
|
||||
fn flushSentence(final_flush: bool) void {
|
||||
const a = allocator;
|
||||
if (sentence_buf.items.len == 0) {
|
||||
if (final_flush) playback.endReply();
|
||||
return;
|
||||
};
|
||||
_ = child.wait(io) catch {};
|
||||
}
|
||||
// Sentences end on . ! ? — hold back a bare trailing terminator-less tail
|
||||
// unless this is the end of the reply.
|
||||
if (!final_flush) {
|
||||
const last = sentence_buf.items[sentence_buf.items.len - 1];
|
||||
if (last != '.' and last != '!' and last != '?') return;
|
||||
}
|
||||
const s = sentence_buf.toOwnedSlice(a) catch return;
|
||||
playback.push(s);
|
||||
if (final_flush) playback.endReply();
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------- capture ---
|
||||
@@ -321,17 +332,10 @@ fn processUtterance() void {
|
||||
defer allocator.free(response);
|
||||
std.debug.print("[INFERON]\n{s}\n\n", .{response});
|
||||
|
||||
// --- tts + playback ---
|
||||
// --- tts + playback: sentences already streaming; flush the tail ---
|
||||
setState(.speaking);
|
||||
var duration: f64 = 0;
|
||||
if (melo.speak(allocator, io, response, &duration)) |tts_wav| {
|
||||
defer allocator.free(tts_wav);
|
||||
std.debug.print("inferon: tts ready {s} ({d:.2}s synth)\n", .{ tts_wav, duration });
|
||||
playWav(tts_wav);
|
||||
cleanupFile(tts_wav);
|
||||
} else |e| {
|
||||
std.debug.print("inferon: tts failed: {s}\n", .{@errorName(e)});
|
||||
}
|
||||
flushSentence(true);
|
||||
playback.waitDrained();
|
||||
|
||||
// After speaking: back to idle-but-in-conversation. User clicks when
|
||||
// they want to talk again.
|
||||
@@ -476,12 +480,20 @@ pub fn main(init: std.process.Init) !void {
|
||||
};
|
||||
if (melo.waitReady(allocator, init.io)) {
|
||||
std.debug.print("inferon: melo ready\n", .{});
|
||||
playback.start(allocator, init.io);
|
||||
} else {
|
||||
std.debug.print("inferon: melo NOT ready\n", .{});
|
||||
}
|
||||
|
||||
_ = QApplication.exec();
|
||||
|
||||
// Session-lifetime buffers: free so the debug allocator doesn't squawk.
|
||||
if (last_user_text) |t| allocator.free(t);
|
||||
last_user_text = null;
|
||||
sentence_buf.deinit(allocator);
|
||||
overlay_mod.Overlay.shutdownMem(); // static, no instance needed
|
||||
playback.shutdown();
|
||||
|
||||
whisper.stop();
|
||||
melo.stop();
|
||||
}
|
||||
Reference in new issue
Block a user