MeloTTS voice output, signal handling, auto pipeline

- melo_server.py subprocess (MeloTTS v3 EN-Newest, speed 1.3): line protocol
  stdin->stdout, model loaded once, warmup before READY
- src/melo.zig: lifecycle ownership (spawn/kill with inferon), line protocol
  client, reply sanitizer (newlines/markdown stripped before synthesis)
- SIGINT/SIGTERM handlers: async-signal-safe kill of whisper/melo/recorder,
  default disposition restored + re-raise for truthful exit status
- processUtterance: transcribe->infer->speak chains automatically after
  stop-click; mid-pipeline clicks are no-ops; back to idle after reply
- temp cleanup: recording WAV deleted after transcription, TTS WAV after
  playback
- whisper: pid() export for signal handler

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
pierreandLetta Code committed 2026-08-25 00:53:48 +03:00
1 parent 3ddb0e77bc
commit 90b59e20f3
4 files changed
+362 -42

No files matched your search

+124 -42
View File
@@ -4,6 +4,7 @@
const std = @import("std");
const Io = std.Io;
const whisper = @import("whisper.zig");
const melo = @import("melo.zig");
const qt6 = @import("libqt6zig");
const QApplication = qt6.QApplication;
const QSystemTrayIcon = qt6.QSystemTrayIcon;
@@ -98,6 +99,22 @@ fn setState(next: State) void {
tray_icon.setToolTip(tip);
}
/// Play a WAV via pw-play (blocking; audio length). Fine for v1 — same
/// thread already blocks on whisper/letta. UI-thread decoupling comes later.
fn playWav(path: []const u8) void {
const io = io_ctx orelse return;
var child = std.process.spawn(io, .{
.argv = &.{ "pw-play", path },
.stdin = .ignore,
.stdout = .ignore,
.stderr = .ignore,
}) catch |e| {
std.debug.print("inferon: pw-play spawn failed: {s}\n", .{@errorName(e)});
return;
};
_ = child.wait(io) catch {};
}
// --------------------------------------------------------------- capture ---
var recorder: ?std.process.Child = null;
@@ -170,11 +187,66 @@ fn stopRecording() void {
fn onQuit(_: QAction) callconv(.c) void {
if (recorder != null) stopRecording();
whisper.stop(); // whisper-server dies with us
melo.stop();
QApplication.quit();
}
// -------------------------------------------------------------- handlers ---
/// Delete a temp file if it exists. Best-effort — /tmp cleanup, not critical.
fn cleanupFile(path: []const u8) void {
const io = io_ctx orelse return;
Io.Dir.deleteFileAbsolute(io, path) catch {};
}
/// Full pipeline after recording stops: transcribe -> infer -> speak.
/// All blocking on the UI thread for now (same as before); states advance
/// automatically, no manual clicking between stages.
fn processUtterance() void {
const io = io_ctx orelse return;
const wav_in = recording_path orelse return;
defer {
cleanupFile(wav_in);
allocator.free(wav_in);
recording_path = null;
}
setState(.transcribing);
const text: []u8 = whisper.transcribe(allocator, io, wav_in) catch |e| {
std.debug.print("inferon: transcription failed: {s}\n", .{@errorName(e)});
setState(.error_backend);
return;
};
std.debug.print("inferon: transcript: \"{s}\"\n", .{text});
setLastUserText(text);
// --- agent ---
setState(.thinking);
const response = letta.infer(io, allocator, LETTA_AGENT_ID, text) catch |err| {
std.log.err("letta failed: {s}", .{@errorName(err)});
setState(.idle);
return;
};
defer allocator.free(response);
std.debug.print("[INFERON]\n{s}\n\n", .{response});
// --- tts + playback ---
setState(.speaking);
var duration: f64 = 0;
if (melo.speak(allocator, io, response, &duration)) |tts_wav| {
defer allocator.free(tts_wav);
std.debug.print("inferon: tts ready {s} ({d:.2}s synth)\n", .{ tts_wav, duration });
playWav(tts_wav);
cleanupFile(tts_wav);
} else |e| {
std.debug.print("inferon: tts failed: {s}\n", .{@errorName(e)});
}
// After speaking: back to idle-but-in-conversation. User clicks when
// they want to talk again.
setState(.idle);
}
fn onTrayActivated(_: QSystemTrayIcon, reason: i32) callconv(.c) void {
// Trigger = 3 (click). Only toggle on left click; context menu handles the rest.
if (reason != 3) return;
@@ -186,49 +258,10 @@ fn onTrayActivated(_: QSystemTrayIcon, reason: i32) callconv(.c) void {
},
.recording => {
stopRecording();
setState(.transcribing);
// TODO: move off UI thread once agent loop lands (blocking call).
if (recording_path) |p| {
const io = io_ctx.?;
if (whisper.transcribe(allocator, io, p)) |text| {
std.debug.print("inferon: transcript: \"{s}\"\n", .{text});
setLastUserText(text);
setState(.thinking);
} else |e| {
std.debug.print("inferon: transcription failed: {s}\n", .{@errorName(e)});
setState(.error_backend);
}
}
processUtterance(); // chains transcribe -> infer -> speak -> re-listen
},
// Placeholder transitions until agent wired in:
.transcribing => setState(.thinking),
.thinking => {
const prompt = last_user_text orelse {
std.debug.print("inferon: no transcript to send\n", .{});
setState(.idle);
return;
};
std.debug.print("Sending to agent: {s}\n", .{prompt});
const response = letta.infer(
io_ctx.?,
allocator,
LETTA_AGENT_ID,
prompt,
) catch |err| {
std.log.err("letta failed: {s}", .{@errorName(err)});
setState(.idle); // adapt to whatever recovery makes sense in your loop
return;
};
defer allocator.free(response);
setState(.speaking);
// TODO: parse `response` (JSON? plain text?) and update UI/state
std.debug.print("[INFERON]\n{s}\n\n", .{response});
},
.speaking => if (conversation_active) startRecording() else setState(.idle),
// Clicks during the pipeline are no-ops now — states advance on their own.
.transcribing, .thinking, .speaking => {},
.error_backend => setState(.idle),
}
}
@@ -241,6 +274,42 @@ fn onEndConversation(_: QAction) callconv(.c) void {
setState(.idle);
}
// --------------------------------------------------------------- signals ---
/// Async-signal-safe teardown: raw kill() of children only, no Io calls.
/// Children are orphan-reaped by init when we exit right after.
const SigType = @TypeOf(std.posix.SIG.INT);
fn handleFatalSignal(sig: SigType) callconv(.c) void {
if (whisper.pid()) |p| {
std.posix.kill(p, std.posix.SIG.TERM) catch {};
}
if (melo.pid()) |p| {
std.posix.kill(p, std.posix.SIG.TERM) catch {};
}
if (recorder) |c| {
if (c.id) |p| std.posix.kill(p, std.posix.SIG.TERM) catch {};
}
// Restore default handler and re-raise so the exit status is truthful.
const act = std.posix.Sigaction{
.handler = .{ .handler = std.posix.SIG.DFL },
.mask = std.posix.sigemptyset(),
.flags = 0,
};
std.posix.sigaction(sig, &act, null);
std.posix.raise(sig) catch {};
}
fn installSignalHandlers() void {
const act = std.posix.Sigaction{
.handler = .{ .handler = handleFatalSignal },
.mask = std.posix.sigemptyset(),
.flags = 0, // no SA_RESTART: let blocking syscalls die too
};
std.posix.sigaction(std.posix.SIG.INT, &act, null);
std.posix.sigaction(std.posix.SIG.TERM, &act, null);
}
// ------------------------------------------------------------------ main ---
pub fn main(init: std.process.Init) !void {
@@ -259,6 +328,8 @@ pub fn main(init: std.process.Init) !void {
allocator = init.gpa;
io_ctx = init.io;
installSignalHandlers();
const quit_action = QAction.new5("&Quit", QWidget{ .ptr = null });
quit_action.onTriggered(onQuit);
@@ -290,7 +361,18 @@ pub fn main(init: std.process.Init) !void {
std.debug.print("inferon: whisper-server NOT ready after {d}s\n", .{whisper.READY_TIMEOUT_S});
}
// TTS backend: melo_server.py (fails soft — notifications still work).
melo.start(allocator, init.io) catch |e| {
std.debug.print("inferon: failed to start melo server: {s}\n", .{@errorName(e)});
};
if (melo.waitReady(allocator, init.io)) {
std.debug.print("inferon: melo ready\n", .{});
} else {
std.debug.print("inferon: melo NOT ready\n", .{});
}
_ = QApplication.exec();
whisper.stop();
melo.stop();
}