MeloTTS voice output, signal handling, auto pipeline
- melo_server.py subprocess (MeloTTS v3 EN-Newest, speed 1.3): line protocol stdin->stdout, model loaded once, warmup before READY - src/melo.zig: lifecycle ownership (spawn/kill with inferon), line protocol client, reply sanitizer (newlines/markdown stripped before synthesis) - SIGINT/SIGTERM handlers: async-signal-safe kill of whisper/melo/recorder, default disposition restored + re-raise for truthful exit status - processUtterance: transcribe->infer->speak chains automatically after stop-click; mid-pipeline clicks are no-ops; back to idle after reply - temp cleanup: recording WAV deleted after transcription, TTS WAV after playback - whisper: pid() export for signal handler 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
1 parent
3ddb0e77bc
commit
90b59e20f3
4 files changed
+362
-42
No files matched your search
+124
-42
@@ -4,6 +4,7 @@
|
||||
const std = @import("std");
|
||||
const Io = std.Io;
|
||||
const whisper = @import("whisper.zig");
|
||||
const melo = @import("melo.zig");
|
||||
const qt6 = @import("libqt6zig");
|
||||
const QApplication = qt6.QApplication;
|
||||
const QSystemTrayIcon = qt6.QSystemTrayIcon;
|
||||
@@ -98,6 +99,22 @@ fn setState(next: State) void {
|
||||
tray_icon.setToolTip(tip);
|
||||
}
|
||||
|
||||
/// Play a WAV via pw-play (blocking; audio length). Fine for v1 — same
|
||||
/// thread already blocks on whisper/letta. UI-thread decoupling comes later.
|
||||
fn playWav(path: []const u8) void {
|
||||
const io = io_ctx orelse return;
|
||||
var child = std.process.spawn(io, .{
|
||||
.argv = &.{ "pw-play", path },
|
||||
.stdin = .ignore,
|
||||
.stdout = .ignore,
|
||||
.stderr = .ignore,
|
||||
}) catch |e| {
|
||||
std.debug.print("inferon: pw-play spawn failed: {s}\n", .{@errorName(e)});
|
||||
return;
|
||||
};
|
||||
_ = child.wait(io) catch {};
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------- capture ---
|
||||
|
||||
var recorder: ?std.process.Child = null;
|
||||
@@ -170,11 +187,66 @@ fn stopRecording() void {
|
||||
fn onQuit(_: QAction) callconv(.c) void {
|
||||
if (recorder != null) stopRecording();
|
||||
whisper.stop(); // whisper-server dies with us
|
||||
melo.stop();
|
||||
QApplication.quit();
|
||||
}
|
||||
|
||||
// -------------------------------------------------------------- handlers ---
|
||||
|
||||
/// Delete a temp file if it exists. Best-effort — /tmp cleanup, not critical.
|
||||
fn cleanupFile(path: []const u8) void {
|
||||
const io = io_ctx orelse return;
|
||||
Io.Dir.deleteFileAbsolute(io, path) catch {};
|
||||
}
|
||||
|
||||
/// Full pipeline after recording stops: transcribe -> infer -> speak.
|
||||
/// All blocking on the UI thread for now (same as before); states advance
|
||||
/// automatically, no manual clicking between stages.
|
||||
fn processUtterance() void {
|
||||
const io = io_ctx orelse return;
|
||||
const wav_in = recording_path orelse return;
|
||||
defer {
|
||||
cleanupFile(wav_in);
|
||||
allocator.free(wav_in);
|
||||
recording_path = null;
|
||||
}
|
||||
|
||||
setState(.transcribing);
|
||||
const text: []u8 = whisper.transcribe(allocator, io, wav_in) catch |e| {
|
||||
std.debug.print("inferon: transcription failed: {s}\n", .{@errorName(e)});
|
||||
setState(.error_backend);
|
||||
return;
|
||||
};
|
||||
std.debug.print("inferon: transcript: \"{s}\"\n", .{text});
|
||||
setLastUserText(text);
|
||||
|
||||
// --- agent ---
|
||||
setState(.thinking);
|
||||
const response = letta.infer(io, allocator, LETTA_AGENT_ID, text) catch |err| {
|
||||
std.log.err("letta failed: {s}", .{@errorName(err)});
|
||||
setState(.idle);
|
||||
return;
|
||||
};
|
||||
defer allocator.free(response);
|
||||
std.debug.print("[INFERON]\n{s}\n\n", .{response});
|
||||
|
||||
// --- tts + playback ---
|
||||
setState(.speaking);
|
||||
var duration: f64 = 0;
|
||||
if (melo.speak(allocator, io, response, &duration)) |tts_wav| {
|
||||
defer allocator.free(tts_wav);
|
||||
std.debug.print("inferon: tts ready {s} ({d:.2}s synth)\n", .{ tts_wav, duration });
|
||||
playWav(tts_wav);
|
||||
cleanupFile(tts_wav);
|
||||
} else |e| {
|
||||
std.debug.print("inferon: tts failed: {s}\n", .{@errorName(e)});
|
||||
}
|
||||
|
||||
// After speaking: back to idle-but-in-conversation. User clicks when
|
||||
// they want to talk again.
|
||||
setState(.idle);
|
||||
}
|
||||
|
||||
fn onTrayActivated(_: QSystemTrayIcon, reason: i32) callconv(.c) void {
|
||||
// Trigger = 3 (click). Only toggle on left click; context menu handles the rest.
|
||||
if (reason != 3) return;
|
||||
@@ -186,49 +258,10 @@ fn onTrayActivated(_: QSystemTrayIcon, reason: i32) callconv(.c) void {
|
||||
},
|
||||
.recording => {
|
||||
stopRecording();
|
||||
setState(.transcribing);
|
||||
// TODO: move off UI thread once agent loop lands (blocking call).
|
||||
if (recording_path) |p| {
|
||||
const io = io_ctx.?;
|
||||
if (whisper.transcribe(allocator, io, p)) |text| {
|
||||
std.debug.print("inferon: transcript: \"{s}\"\n", .{text});
|
||||
setLastUserText(text);
|
||||
setState(.thinking);
|
||||
} else |e| {
|
||||
std.debug.print("inferon: transcription failed: {s}\n", .{@errorName(e)});
|
||||
setState(.error_backend);
|
||||
}
|
||||
}
|
||||
processUtterance(); // chains transcribe -> infer -> speak -> re-listen
|
||||
},
|
||||
// Placeholder transitions until agent wired in:
|
||||
.transcribing => setState(.thinking),
|
||||
.thinking => {
|
||||
const prompt = last_user_text orelse {
|
||||
std.debug.print("inferon: no transcript to send\n", .{});
|
||||
setState(.idle);
|
||||
return;
|
||||
};
|
||||
|
||||
std.debug.print("Sending to agent: {s}\n", .{prompt});
|
||||
|
||||
const response = letta.infer(
|
||||
io_ctx.?,
|
||||
allocator,
|
||||
LETTA_AGENT_ID,
|
||||
prompt,
|
||||
) catch |err| {
|
||||
std.log.err("letta failed: {s}", .{@errorName(err)});
|
||||
setState(.idle); // adapt to whatever recovery makes sense in your loop
|
||||
return;
|
||||
};
|
||||
defer allocator.free(response);
|
||||
|
||||
setState(.speaking);
|
||||
|
||||
// TODO: parse `response` (JSON? plain text?) and update UI/state
|
||||
std.debug.print("[INFERON]\n{s}\n\n", .{response});
|
||||
},
|
||||
.speaking => if (conversation_active) startRecording() else setState(.idle),
|
||||
// Clicks during the pipeline are no-ops now — states advance on their own.
|
||||
.transcribing, .thinking, .speaking => {},
|
||||
.error_backend => setState(.idle),
|
||||
}
|
||||
}
|
||||
@@ -241,6 +274,42 @@ fn onEndConversation(_: QAction) callconv(.c) void {
|
||||
setState(.idle);
|
||||
}
|
||||
|
||||
// --------------------------------------------------------------- signals ---
|
||||
|
||||
/// Async-signal-safe teardown: raw kill() of children only, no Io calls.
|
||||
/// Children are orphan-reaped by init when we exit right after.
|
||||
const SigType = @TypeOf(std.posix.SIG.INT);
|
||||
|
||||
fn handleFatalSignal(sig: SigType) callconv(.c) void {
|
||||
if (whisper.pid()) |p| {
|
||||
std.posix.kill(p, std.posix.SIG.TERM) catch {};
|
||||
}
|
||||
if (melo.pid()) |p| {
|
||||
std.posix.kill(p, std.posix.SIG.TERM) catch {};
|
||||
}
|
||||
if (recorder) |c| {
|
||||
if (c.id) |p| std.posix.kill(p, std.posix.SIG.TERM) catch {};
|
||||
}
|
||||
// Restore default handler and re-raise so the exit status is truthful.
|
||||
const act = std.posix.Sigaction{
|
||||
.handler = .{ .handler = std.posix.SIG.DFL },
|
||||
.mask = std.posix.sigemptyset(),
|
||||
.flags = 0,
|
||||
};
|
||||
std.posix.sigaction(sig, &act, null);
|
||||
std.posix.raise(sig) catch {};
|
||||
}
|
||||
|
||||
fn installSignalHandlers() void {
|
||||
const act = std.posix.Sigaction{
|
||||
.handler = .{ .handler = handleFatalSignal },
|
||||
.mask = std.posix.sigemptyset(),
|
||||
.flags = 0, // no SA_RESTART: let blocking syscalls die too
|
||||
};
|
||||
std.posix.sigaction(std.posix.SIG.INT, &act, null);
|
||||
std.posix.sigaction(std.posix.SIG.TERM, &act, null);
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------ main ---
|
||||
|
||||
pub fn main(init: std.process.Init) !void {
|
||||
@@ -259,6 +328,8 @@ pub fn main(init: std.process.Init) !void {
|
||||
allocator = init.gpa;
|
||||
io_ctx = init.io;
|
||||
|
||||
installSignalHandlers();
|
||||
|
||||
const quit_action = QAction.new5("&Quit", QWidget{ .ptr = null });
|
||||
quit_action.onTriggered(onQuit);
|
||||
|
||||
@@ -290,7 +361,18 @@ pub fn main(init: std.process.Init) !void {
|
||||
std.debug.print("inferon: whisper-server NOT ready after {d}s\n", .{whisper.READY_TIMEOUT_S});
|
||||
}
|
||||
|
||||
// TTS backend: melo_server.py (fails soft — notifications still work).
|
||||
melo.start(allocator, init.io) catch |e| {
|
||||
std.debug.print("inferon: failed to start melo server: {s}\n", .{@errorName(e)});
|
||||
};
|
||||
if (melo.waitReady(allocator, init.io)) {
|
||||
std.debug.print("inferon: melo ready\n", .{});
|
||||
} else {
|
||||
std.debug.print("inferon: melo NOT ready\n", .{});
|
||||
}
|
||||
|
||||
_ = QApplication.exec();
|
||||
|
||||
whisper.stop();
|
||||
melo.stop();
|
||||
}
|
||||
Reference in new issue
Block a user