Streaming overlay: chat GUI for agent responses
- overlay.zig: frameless translucent panel (fixed bg + rounded corners, only bubbles scroll), drag handle, hover-to-full-opacity, top-right spawn - chat layout: per-message QLabel bubbles (user purple / agent green / tool steps grey mono), transcript retained across turns, End-conversation resets - agent reply streams into its own growing bubble (re-render per chunk) - basic markdown in all messages: **bold**, *italic*, `code`, HTML-escaped - tool steps truncated to 90 chars, click to expand/collapse - polite autoscroll: pins to bottom unless user scrolled up - letta.zig: streamInfer() — --output-format stream-json, per-event callback (tool calls, returns, reply chunks), full reply still returned for TTS - whisper: --language auto (no more hardcoded English) - fix: stderr_log leak crashing Quit via debug allocator 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
1 parent
5b050c09e3
commit
2b2235ff86
5 files changed
+500
-1
No files matched your search
+127
@@ -41,3 +41,130 @@ pub fn infer(
|
||||
|
||||
return output.toOwnedSlice(allocator);
|
||||
}
|
||||
|
||||
|
||||
// ------------------------------------------------------------- streaming ---
|
||||
|
||||
pub const Event = union(enum) {
|
||||
step: []const u8, // tool call / return, pre-formatted line
|
||||
chunk: []const u8, // streamed reply text
|
||||
};
|
||||
|
||||
/// Like infer(), but spawns letta with --output-format stream-json and calls
|
||||
/// `onEvent` (on THIS thread — caller marshals to Qt) for every step/chunk.
|
||||
/// Returns the final full reply text (allocated, caller frees).
|
||||
pub fn streamInfer(
|
||||
io: std.Io,
|
||||
allocator: std.mem.Allocator,
|
||||
agent: []const u8,
|
||||
prompt: []const u8,
|
||||
ctx: ?*anyopaque,
|
||||
onEvent: *const fn (ctx: ?*anyopaque, ev: Event) void,
|
||||
) ![]u8 {
|
||||
var child = try std.process.spawn(io, .{
|
||||
.argv = &.{
|
||||
LETTA_PATH, "--agent", agent,
|
||||
"-p", prompt, "--conversation",
|
||||
"default", "--toolset", "default",
|
||||
"--output-format", "stream-json",
|
||||
},
|
||||
.stdout = .pipe,
|
||||
.stderr = .inherit,
|
||||
});
|
||||
const stdout_file = child.stdout.?;
|
||||
|
||||
var reply: std.ArrayList(u8) = .empty;
|
||||
errdefer reply.deinit(allocator);
|
||||
|
||||
var line_buf: [16 * 1024]u8 = undefined;
|
||||
var line_len: usize = 0;
|
||||
var buf: [16 * 1024]u8 = undefined;
|
||||
|
||||
while (true) {
|
||||
const n = stdout_file.readStreaming(io, &.{&buf}) catch break;
|
||||
if (n == 0) break;
|
||||
for (buf[0..n]) |ch| {
|
||||
if (ch == '\n') {
|
||||
if (line_len > 0) {
|
||||
handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply);
|
||||
line_len = 0;
|
||||
}
|
||||
} else if (line_len < line_buf.len) {
|
||||
line_buf[line_len] = ch;
|
||||
line_len += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
_ = child.wait(io) catch {};
|
||||
|
||||
return reply.toOwnedSlice(allocator);
|
||||
}
|
||||
|
||||
/// Minimal hand parse of one stream-json line (we control the producer).
|
||||
fn handleLine(
|
||||
allocator: std.mem.Allocator,
|
||||
line: []const u8,
|
||||
ctx: ?*anyopaque,
|
||||
onEvent: *const fn (ctx: ?*anyopaque, ev: Event) void,
|
||||
reply: *std.ArrayList(u8),
|
||||
) void {
|
||||
const mt = dupeStr(allocator, line, "message_type") orelse {
|
||||
// Non-message line; the "result" line holds the full text as backup.
|
||||
if (std.mem.indexOf(u8, line, "\"type\":\"result\"") != null and reply.items.len == 0) {
|
||||
if (dupeStr(allocator, line, "result")) |r| {
|
||||
defer allocator.free(r);
|
||||
reply.appendSlice(allocator, r) catch {};
|
||||
}
|
||||
}
|
||||
return;
|
||||
};
|
||||
defer allocator.free(mt);
|
||||
|
||||
if (std.mem.eql(u8, mt, "tool_call_message")) {
|
||||
const name = dupeStr(allocator, line, "name") orelse allocator.dupe(u8, "?") catch return;
|
||||
defer allocator.free(name);
|
||||
const args = dupeStr(allocator, line, "command") orelse
|
||||
dupeStr(allocator, line, "description") orelse allocator.dupe(u8, "") catch return;
|
||||
defer allocator.free(args);
|
||||
const text = std.fmt.allocPrint(allocator, "> {s} {s}", .{ name, args }) catch return;
|
||||
defer allocator.free(text);
|
||||
onEvent(ctx, .{ .step = text });
|
||||
} else if (std.mem.eql(u8, mt, "tool_return_message")) {
|
||||
const ret = dupeStr(allocator, line, "tool_return") orelse allocator.dupe(u8, "") catch return;
|
||||
defer allocator.free(ret);
|
||||
const text = std.fmt.allocPrint(allocator, " {s}", .{ret}) catch return;
|
||||
defer allocator.free(text);
|
||||
onEvent(ctx, .{ .step = text });
|
||||
} else if (std.mem.eql(u8, mt, "assistant_message")) {
|
||||
const txt = dupeStr(allocator, line, "text") orelse return;
|
||||
defer allocator.free(txt);
|
||||
reply.appendSlice(allocator, txt) catch {};
|
||||
onEvent(ctx, .{ .chunk = txt });
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract "key":"value" (with escape handling), allocated with `allocator`.
|
||||
fn dupeStr(allocator: std.mem.Allocator, line: []const u8, key: []const u8) ?[]u8 {
|
||||
var pat_buf: [64]u8 = undefined;
|
||||
const pat = std.fmt.bufPrint(&pat_buf, "\"{s}\":\"", .{key}) catch return null;
|
||||
const start = std.mem.indexOf(u8, line, pat) orelse return null;
|
||||
var it = line[start + pat.len ..];
|
||||
|
||||
var out: std.ArrayList(u8) = .empty;
|
||||
errdefer out.deinit(allocator);
|
||||
while (it.len > 0 and it[0] != '"') {
|
||||
if (it[0] == '\\' and it.len > 1) {
|
||||
const esc: u8 = switch (it[1]) {
|
||||
'n' => '\n',
|
||||
't' => '\t',
|
||||
else => it[1],
|
||||
};
|
||||
out.append(allocator, esc) catch return null;
|
||||
it = it[2..];
|
||||
} else {
|
||||
out.append(allocator, it[0]) catch return null;
|
||||
it = it[1..];
|
||||
}
|
||||
}
|
||||
return out.toOwnedSlice(allocator) catch null;
|
||||
}
|
||||
Reference in new issue
Block a user