From 27327e94b0913664c802962bcd10be910eb210ec Mon Sep 17 00:00:00 2001 From: Pierre De Lancre Date: Mon, 21 Sep 2026 06:02:59 +0300 Subject: [PATCH] =?UTF-8?q?letta:=20fix=20truncated=20replies=20=E2=80=94?= =?UTF-8?q?=20process=20final=20unterminated=20line=20at=20EOF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The stream reader only handled lines terminated by newline. The letta CLI final JSON line arrives without trailing newline at EOF, so the last assistant message (or result line) sat in the line buffer and was silently dropped — replies ended mid-sentence. Also: read errors no longer silently end collection (logged), and stream completeness is tracked via the result line with a warning when absent. 👾 Generated with [Letta Code](https://letta.com) Co-Authored-By: Letta Code --- src/letta.zig | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/src/letta.zig b/src/letta.zig index 9b7a8b8..8a9a5bc 100644 --- a/src/letta.zig +++ b/src/letta.zig @@ -120,8 +120,12 @@ pub fn streamInferConv( var line_len: usize = 0; var buf: [16 * 1024]u8 = undefined; + var stream_ok = false; // did we see a proper result/completion line? while (true) { - const n = stdout_file.readStreaming(io, &.{&buf}) catch break; + const n = stdout_file.readStreaming(io, &.{&buf}) catch |e| { + std.debug.print("inferon/letta: stdout read error {s} — reply may be TRUNCATED\n", .{@errorName(e)}); + break; + }; if (n == 0) break; raw_tail.appendSlice(allocator, buf[0..n]) catch {}; if (raw_tail.items.len > 4096) { @@ -133,6 +137,7 @@ pub fn streamInferConv( if (ch == '\n') { if (line_len > 0) { handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply); + if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true; line_len = 0; } } else if (line_len < line_buf.len) { @@ -141,8 +146,21 @@ pub fn streamInferConv( } } } + // CRITICAL: the last line may arrive WITHOUT a trailing newline (EOF + // right after the final JSON). It is still a complete line — dropping + // it silently truncated replies (the text of the last assistant message + // or the result line was lost). + if (line_len > 0) { + handleLine(allocator, line_buf[0..line_len], ctx, onEvent, &reply); + if (std.mem.indexOf(u8, line_buf[0..line_len], "\"type\":\"result\"") != null) stream_ok = true; + } _ = child.wait(io) catch {}; + if (!stream_ok) { + std.debug.print("inferon/letta: stream ended WITHOUT result line — reply likely TRUNCATED ({d} bytes collected). tail:\n{s}\n", .{ reply.items.len, raw_tail.items }); + } + std.debug.print("inferon/letta: stream complete, reply {d} bytes, result_line={}\n", .{ reply.items.len, stream_ok }); + if (reply.items.len == 0 and raw_tail.items.len > 0) { std.debug.print("inferon/letta: EMPTY reply, raw stream tail:\n{s}\n", .{raw_tail.items}); }