letta: fix truncated replies — process final unterminated line at EOF

The stream reader only handled lines terminated by newline. The letta
CLI final JSON line arrives without trailing newline at EOF, so the last
assistant message (or result line) sat in the line buffer and was
silently dropped — replies ended mid-sentence. Also: read errors no
longer silently end collection (logged), and stream completeness is
tracked via the result line with a warning when absent.

👾 Generated with [Letta Code](https://letta.com)

Co-Authored-By: Letta Code <noreply@letta.com>
This commit is contained in:
pierreandLetta Code committed 2026-09-21 06:02:59 +03:00
1 parent 81fcbabefc
commit 27327e94b0
1 file changed
+19 -1
+19 -1
View File
@@ -120,8 +120,12 @@ pub fn streamInferConv(
var line_len: usize = 0; var line_len: usize = 0;
var buf: [16 * 1024]u8 = undefined; var buf: [16 * 1024]u8 = undefined;
var stream_ok = false; // did we see a proper result/completion line?
while (true) { while (true) {
const n = stdout_file.readStreaming(io, &.{&buf}) catch break; const n = stdout_file.readStreaming(io, &.{&buf}) catch |e| {
std.debug.print("inferon/letta: stdout read error {s} — reply may be TRUNCATED\n", .{@errorName(e)});
break;
};
if (n == 0) break; if (n == 0) break;
raw_tail.appendSlice(allocator, buf[0..n]) catch {}; raw_tail.appendSlice(allocator, buf[0..n]) catch {};
if (raw_tail.items.len > 4096) { if (raw_tail.items.len > 4096) {
@@ -133,6 +137,7 @@ pub fn streamInferConv(
if (ch == '\n') { if (ch == '\n') {
if (line_len > 0) { if (line_len > 0) {
handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply); handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply);
if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true;
line_len = 0; line_len = 0;
} }
} else if (line_len < line_buf.len) { } else if (line_len < line_buf.len) {
@@ -141,8 +146,21 @@ pub fn streamInferConv(
} }
} }
} }
// CRITICAL: the last line may arrive WITHOUT a trailing newline (EOF
// right after the final JSON). It is still a complete line — dropping
// it silently truncated replies (the text of the last assistant message
// or the result line was lost).
if (line_len > 0) {
handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply);
if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true;
}
_ = child.wait(io) catch {}; _ = child.wait(io) catch {};
if (!stream_ok) {
std.debug.print("inferon/letta: stream ended WITHOUT result line — reply likely TRUNCATED ({d} bytes collected). tail:\n{s}\n", .{ reply.items.len, raw_tail.items });
}
std.debug.print("inferon/letta: stream complete, reply {d} bytes, result_line={}\n", .{ reply.items.len, stream_ok });
if (reply.items.len == 0 and raw_tail.items.len > 0) { if (reply.items.len == 0 and raw_tail.items.len > 0) {
std.debug.print("inferon/letta: EMPTY reply, raw stream tail:\n{s}\n", .{raw_tail.items}); std.debug.print("inferon/letta: EMPTY reply, raw stream tail:\n{s}\n", .{raw_tail.items});
} }