Streaming overlay: chat GUI for agent responses

- overlay.zig: frameless translucent panel (fixed bg + rounded corners,
  only bubbles scroll), drag handle, hover-to-full-opacity, top-right spawn
- chat layout: per-message QLabel bubbles (user purple / agent green /
  tool steps grey mono), transcript retained across turns, End-conversation
  resets
- agent reply streams into its own growing bubble (re-render per chunk)
- basic markdown in all messages: **bold**, *italic*, `code`, HTML-escaped
- tool steps truncated to 90 chars, click to expand/collapse
- polite autoscroll: pins to bottom unless user scrolled up
- letta.zig: streamInfer() — --output-format stream-json, per-event callback
  (tool calls, returns, reply chunks), full reply still returned for TTS
- whisper: --language auto (no more hardcoded English)
- fix: stderr_log leak crashing Quit via debug allocator

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
pierreandLetta Code committed 2026-09-01 20:37:45 +03:00
1 parent 5b050c09e3
commit 2b2235ff86
5 files changed
+500 -1

No files matched your search

+127
View File
@@ -41,3 +41,130 @@ pub fn infer(
return output.toOwnedSlice(allocator);
}
// ------------------------------------------------------------- streaming ---
pub const Event = union(enum) {
step: []const u8, // tool call / return, pre-formatted line
chunk: []const u8, // streamed reply text
};
/// Like infer(), but spawns letta with --output-format stream-json and calls
/// `onEvent` (on THIS thread — caller marshals to Qt) for every step/chunk.
/// Returns the final full reply text (allocated, caller frees).
pub fn streamInfer(
io: std.Io,
allocator: std.mem.Allocator,
agent: []const u8,
prompt: []const u8,
ctx: ?*anyopaque,
onEvent: *const fn (ctx: ?*anyopaque, ev: Event) void,
) ![]u8 {
var child = try std.process.spawn(io, .{
.argv = &.{
LETTA_PATH, "--agent", agent,
"-p", prompt, "--conversation",
"default", "--toolset", "default",
"--output-format", "stream-json",
},
.stdout = .pipe,
.stderr = .inherit,
});
const stdout_file = child.stdout.?;
var reply: std.ArrayList(u8) = .empty;
errdefer reply.deinit(allocator);
var line_buf: [16 * 1024]u8 = undefined;
var line_len: usize = 0;
var buf: [16 * 1024]u8 = undefined;
while (true) {
const n = stdout_file.readStreaming(io, &.{&buf}) catch break;
if (n == 0) break;
for (buf[0..n]) |ch| {
if (ch == '\n') {
if (line_len > 0) {
handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply);
line_len = 0;
}
} else if (line_len < line_buf.len) {
line_buf[line_len] = ch;
line_len += 1;
}
}
}
_ = child.wait(io) catch {};
return reply.toOwnedSlice(allocator);
}
/// Minimal hand parse of one stream-json line (we control the producer).
fn handleLine(
allocator: std.mem.Allocator,
line: []const u8,
ctx: ?*anyopaque,
onEvent: *const fn (ctx: ?*anyopaque, ev: Event) void,
reply: *std.ArrayList(u8),
) void {
const mt = dupeStr(allocator, line, "message_type") orelse {
// Non-message line; the "result" line holds the full text as backup.
if (std.mem.indexOf(u8, line, "\"type\":\"result\"") != null and reply.items.len == 0) {
if (dupeStr(allocator, line, "result")) |r| {
defer allocator.free(r);
reply.appendSlice(allocator, r) catch {};
}
}
return;
};
defer allocator.free(mt);
if (std.mem.eql(u8, mt, "tool_call_message")) {
const name = dupeStr(allocator, line, "name") orelse allocator.dupe(u8, "?") catch return;
defer allocator.free(name);
const args = dupeStr(allocator, line, "command") orelse
dupeStr(allocator, line, "description") orelse allocator.dupe(u8, "") catch return;
defer allocator.free(args);
const text = std.fmt.allocPrint(allocator, "> {s} {s}", .{ name, args }) catch return;
defer allocator.free(text);
onEvent(ctx, .{ .step = text });
} else if (std.mem.eql(u8, mt, "tool_return_message")) {
const ret = dupeStr(allocator, line, "tool_return") orelse allocator.dupe(u8, "") catch return;
defer allocator.free(ret);
const text = std.fmt.allocPrint(allocator, " {s}", .{ret}) catch return;
defer allocator.free(text);
onEvent(ctx, .{ .step = text });
} else if (std.mem.eql(u8, mt, "assistant_message")) {
const txt = dupeStr(allocator, line, "text") orelse return;
defer allocator.free(txt);
reply.appendSlice(allocator, txt) catch {};
onEvent(ctx, .{ .chunk = txt });
}
}
/// Extract "key":"value" (with escape handling), allocated with `allocator`.
fn dupeStr(allocator: std.mem.Allocator, line: []const u8, key: []const u8) ?[]u8 {
var pat_buf: [64]u8 = undefined;
const pat = std.fmt.bufPrint(&pat_buf, "\"{s}\":\"", .{key}) catch return null;
const start = std.mem.indexOf(u8, line, pat) orelse return null;
var it = line[start + pat.len ..];
var out: std.ArrayList(u8) = .empty;
errdefer out.deinit(allocator);
while (it.len > 0 and it[0] != '"') {
if (it[0] == '\\' and it.len > 1) {
const esc: u8 = switch (it[1]) {
'n' => '\n',
't' => '\t',
else => it[1],
};
out.append(allocator, esc) catch return null;
it = it[2..];
} else {
out.append(allocator, it[0]) catch return null;
it = it[1..];
}
}
return out.toOwnedSlice(allocator) catch null;
}